added potion-polyglot worker folder w/o repos

This commit is contained in:
2026-09-08 21:58:19 -04:00
parent f10303b8c2
commit a16a457669
211 changed files with 41614 additions and 0 deletions

View File

@@ -0,0 +1,436 @@
"""Convert seed conversations between harness-native session formats, via ATIF.
Parsers turn a native session into ATIF; renderers turn ATIF back into a native
session. Adding a harness is one parser plus one renderer.
Renderers flatten tool calls to narration (`[ran Bash: {...}]` / `[result: ...]`)
rather than rebuilding native tool-call records. Every renderer must flatten
identically — see flatten_steps.
"""
from __future__ import annotations
import json
from typing import Any
ATIF_SCHEMA_VERSION = "ATIF-v1.7"
# Codex rollout record types, used to tell the formats apart.
_CODEX_ROLLOUT_TYPES = frozenset(
{"session_meta", "response_item", "event_msg", "turn_context", "compacted"}
)
# ---------------------------------------------------------------------------
# Format detection
# ---------------------------------------------------------------------------
def detect_format(text: str) -> str | None:
"""Return 'atif', 'claude', 'codex', or None for an unrecognised/empty blob."""
stripped = text.strip()
if not stripped:
return None
# ATIF is a single JSON object, not JSONL.
if stripped.startswith("{") and '"steps"' in stripped:
try:
doc = json.loads(stripped)
except (json.JSONDecodeError, ValueError):
doc = None
if isinstance(doc, dict) and isinstance(doc.get("steps"), list):
return "atif"
for raw in stripped.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
if rec.get("type") in _CODEX_ROLLOUT_TYPES and "message" not in rec:
return "codex"
if rec.get("type") in ("user", "assistant") or "message" in rec:
return "claude"
return None
# ---------------------------------------------------------------------------
# ATIF construction helpers
# ---------------------------------------------------------------------------
def _trajectory(steps: list[dict], *, session_id: str | None = None) -> dict:
return {
"schema_version": ATIF_SCHEMA_VERSION,
"session_id": session_id,
"agent": {"name": "unknown"},
"steps": steps,
}
def _step(
step_id: int,
source: str,
*,
message: str = "",
reasoning: str | None = None,
tool_calls: list[dict] | None = None,
observations: list[dict] | None = None,
timestamp: str | None = None,
) -> dict:
step: dict[str, Any] = {
"step_id": step_id,
"source": source,
"message": message,
"is_copied_context": True,
}
if timestamp:
step["timestamp"] = timestamp
if reasoning:
step["reasoning_content"] = reasoning
if tool_calls:
step["tool_calls"] = tool_calls
if observations:
step["observation"] = {"results": observations}
return step
def _content_text(content: Any) -> str:
"""Text of an ATIF message or a ContentPart list."""
if isinstance(content, str):
return content
if isinstance(content, list):
return "".join(
part.get("text") or "" for part in content if isinstance(part, dict)
)
return ""
# ---------------------------------------------------------------------------
# Parsers: native -> ATIF
# ---------------------------------------------------------------------------
def claude_session_to_atif(jsonl_text: str) -> dict:
"""Parse a Claude Code session.jsonl into ATIF steps.
Claude records tool results on `user` records; they become observations.
"""
steps: list[dict] = []
session_id: str | None = None
for raw in jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
session_id = session_id or rec.get("sessionId")
msg = rec.get("message") or {}
role = msg.get("role") or rec.get("type")
if role not in ("user", "assistant"):
continue
content = msg.get("content")
if content is None:
continue
source = "user" if role == "user" else "agent"
if isinstance(content, str):
steps.append(
_step(
len(steps) + 1, source, message=content, timestamp=rec.get("timestamp")
)
)
continue
text_parts: list[str] = []
reasoning_parts: list[str] = []
tool_calls: list[dict] = []
observations: list[dict] = []
for block in content:
if not isinstance(block, dict):
text_parts.append(str(block))
continue
btype = block.get("type")
if btype == "text":
text_parts.append(block.get("text") or "")
elif btype == "thinking":
reasoning_parts.append(block.get("thinking") or "")
elif btype == "tool_use":
tool_calls.append(
{
"tool_call_id": block.get("id") or f"call_{len(tool_calls) + 1}",
"function_name": block.get("name") or "tool",
"arguments": block.get("input") or {},
}
)
elif btype == "tool_result":
observations.append(
{
"source_call_id": block.get("tool_use_id"),
"content": _content_text(block.get("content")),
}
)
steps.append(
_step(
len(steps) + 1,
source,
message="".join(text_parts),
reasoning="".join(reasoning_parts) or None,
tool_calls=tool_calls or None,
observations=observations or None,
timestamp=rec.get("timestamp"),
)
)
return _trajectory(steps, session_id=session_id)
def codex_rollout_to_atif(jsonl_text: str) -> dict:
"""Parse a codex rollout JSONL into ATIF steps."""
steps: list[dict] = []
session_id: str | None = None
for raw in jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
rtype = rec.get("type")
payload = rec.get("payload") or {}
if rtype == "session_meta":
session_id = session_id or payload.get("id")
continue
if rtype != "response_item":
continue
ptype = payload.get("type")
timestamp = rec.get("timestamp")
if ptype == "message":
role = payload.get("role")
if role not in ("user", "assistant"):
continue
steps.append(
_step(
len(steps) + 1,
"user" if role == "user" else "agent",
message=_content_text(payload.get("content")),
timestamp=timestamp,
)
)
elif ptype in ("function_call", "local_shell_call", "custom_tool_call"):
arguments = payload.get("arguments")
if isinstance(arguments, str):
try:
arguments = json.loads(arguments)
except (json.JSONDecodeError, ValueError):
arguments = {"raw": arguments}
steps.append(
_step(
len(steps) + 1,
"agent",
tool_calls=[
{
"tool_call_id": payload.get("call_id") or f"call_{len(steps)}",
"function_name": payload.get("name") or "tool",
"arguments": arguments or {},
}
],
timestamp=timestamp,
)
)
elif ptype in ("function_call_output", "custom_tool_call_output"):
output = payload.get("output")
if isinstance(output, dict):
output = output.get("content") or json.dumps(output, ensure_ascii=False)
steps.append(
_step(
len(steps) + 1,
"agent",
observations=[
{
"source_call_id": payload.get("call_id"),
"content": output if isinstance(output, str) else "",
}
],
timestamp=timestamp,
)
)
elif ptype == "reasoning":
summary = payload.get("summary")
text = ""
if isinstance(summary, list):
text = "".join(
s.get("text") or "" for s in summary if isinstance(s, dict)
)
if text:
steps.append(_step(len(steps) + 1, "agent", reasoning=text, timestamp=timestamp))
return _trajectory(steps, session_id=session_id)
def to_atif(text: str) -> dict:
"""Parse whichever native format `text` is into ATIF."""
fmt = detect_format(text)
if fmt == "atif":
return json.loads(text)
if fmt == "claude":
return claude_session_to_atif(text)
if fmt == "codex":
return codex_rollout_to_atif(text)
raise ValueError("unrecognised session format (not ATIF, Claude JSONL, or codex rollout)")
# ---------------------------------------------------------------------------
# Flattening — shared by every renderer so the loss stays symmetric
# ---------------------------------------------------------------------------
def flatten_steps(atif: dict) -> list[tuple[str, str]]:
"""ATIF steps -> ordered (role, text) pairs, where role is 'user' or 'agent'.
Tool calls and observations become agent narration; `reasoning_content` is dropped.
"""
out: list[tuple[str, str]] = []
for step in atif.get("steps") or []:
if not isinstance(step, dict):
continue
role = "user" if step.get("source") == "user" else "agent"
text = _content_text(step.get("message"))
if text:
out.append((role, text))
for call in step.get("tool_calls") or []:
if not isinstance(call, dict):
continue
args = json.dumps(call.get("arguments") or {}, ensure_ascii=False)
out.append(("agent", f"[ran {call.get('function_name') or 'tool'}: {args}]"))
observation = step.get("observation") or {}
for result in observation.get("results") or []:
if not isinstance(result, dict):
continue
out.append(("agent", f"[result: {_content_text(result.get('content'))}]"))
return out
# ---------------------------------------------------------------------------
# Renderers: ATIF -> native
# ---------------------------------------------------------------------------
def atif_to_codex_rollout(
atif: dict,
iso_ts: str,
*,
session_meta: dict,
max_total: int | None = None,
) -> list[str]:
"""Render ATIF as codex rollout JSONL that `codex exec resume` can continue.
`session_meta` is supplied by the caller so this module reads no files.
"""
lines = [json.dumps(session_meta)]
budget = float("inf") if max_total is None else max_total
for role, text in flatten_steps(atif):
text = (text or "").strip()
if not text or budget <= 0:
continue
text = text[: int(min(budget, len(text)))]
ctype = "input_text" if role == "user" else "output_text"
lines.append(
json.dumps(
{
"timestamp": iso_ts,
"type": "response_item",
"payload": {
"type": "message",
"role": "user" if role == "user" else "assistant",
"content": [{"type": ctype, "text": text}],
},
}
)
)
budget -= len(text)
return lines
def atif_to_claude_session(
atif: dict,
*,
session_id: str,
cwd: str = "/workspace",
git_branch: str = "main",
version: str = "2.1.87",
iso_ts: str,
max_total: int | None = None,
) -> list[str]:
"""Render ATIF as Claude Code session.jsonl that `claude --resume` can continue.
Records are chained by parentUuid: Claude resumes by walking that chain, not by
file order.
"""
lines: list[str] = []
parent_uuid: str | None = None
budget = float("inf") if max_total is None else max_total
for index, (role, text) in enumerate(flatten_steps(atif), start=1):
text = (text or "").strip()
if not text or budget <= 0:
continue
text = text[: int(min(budget, len(text)))]
uuid = _deterministic_uuid(session_id, index)
claude_role = "user" if role == "user" else "assistant"
record: dict[str, Any] = {
"parentUuid": parent_uuid,
"isSidechain": False,
"userType": "external",
"cwd": cwd,
"sessionId": session_id,
"version": version,
"gitBranch": git_branch,
"type": claude_role,
"uuid": uuid,
"timestamp": iso_ts,
}
if claude_role == "user":
record["message"] = {"role": "user", "content": text}
else:
record["message"] = {
"role": "assistant",
"content": [{"type": "text", "text": text}],
"stop_reason": "end_turn",
}
lines.append(json.dumps(record))
parent_uuid = uuid
budget -= len(text)
return lines
def _deterministic_uuid(session_id: str, index: int) -> str:
"""A stable uuid5 per (session, position), so re-rendering is byte-identical."""
import uuid as _uuid
return str(_uuid.uuid5(_uuid.NAMESPACE_URL, f"raccoon-seed/{session_id}/{index}"))

View File

@@ -0,0 +1,53 @@
"""Shared browser-capability disclosure for the agent harnesses.
Only images for browser-facing repos ship Playwright, so the note is conditional on probing
the sandbox for the `pw` wrapper rather than on anything about the task. Probing keeps the
claim true by construction: telling an agent it has a browser it does not have sends it after
a missing binary. To check an image yourself: `command -v pw`.
Both harnesses disclose the same text through their own mechanism:
- Claude Code: appended to --append-system-prompt (scripts/snapshot_agent.py)
- codex: -c developer_instructions=... (scripts/codex_agent.py), which prepends a
developer message and LEAVES codex's base instructions intact. Verified with
`codex debug prompt-input`. Do not switch to model_instructions_file — that
REPLACES the base instructions.
This module exists so the probe and the text live in one place; a copy in each adapter would
drift and the drift would be invisible (both would still run, just disclosing differently).
"""
from __future__ import annotations
import logging
from pathlib import Path
_log = logging.getLogger(__name__)
_NOTE_FILE = Path(__file__).resolve().parent / "toolset_note_browser.md"
_PROBE = "command -v pw >/dev/null 2>&1 && echo yes || echo no"
def browser_note() -> str:
"""The disclosure text, or "" if the note file is missing (never fatal)."""
try:
return _NOTE_FILE.read_text(encoding="utf-8").strip()
except OSError:
_log.warning("%s missing; browser note omitted", _NOTE_FILE.name)
return ""
async def probe_browser(environment) -> bool:
"""True when this image ships the `pw` wrapper. Best-effort: a failed probe means no
note, never a failed run."""
try:
result = await environment.exec(command=_PROBE, timeout_sec=30)
except Exception as exc:
_log.warning("browser probe failed (%s); omitting the browser note", exc)
return False
# Exact tail match, not a substring: several harbor environments exec through a LOGIN
# shell, whose profile scripts can print to stdout. A banner containing "yes" would
# otherwise claim a browser that isn't there — the precise failure this module exists
# to prevent.
found = (getattr(result, "stdout", "") or "").strip().endswith("yes")
_log.info("browser probe: pw %s", "present" if found else "absent")
return found

View File

@@ -0,0 +1,325 @@
#!/bin/bash
# Build a task's workspace from the local repo.
#
# Usage: scripts/build-workspace.sh <task-slug> [commit]
# Example: scripts/build-workspace.sh my-cool-task 3af4366a6
#
# If commit is omitted, reads it from the task's task.toml.
set -euo pipefail
TOOLKIT_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
REPO_DIR="$TOOLKIT_ROOT/repo"
TASK_SLUG="$1"
TASK_DIR="$TOOLKIT_ROOT/harbor-tasks/$TASK_SLUG"
if [ ! -d "$TASK_DIR" ]; then
echo "Error: task directory not found at $TASK_DIR" >&2
exit 1
fi
# The member this task targets, per task.toml ([metadata].repo). Used to resolve both
# the source repo (polyglot) and the member's deterministic checks (below).
# `|| true` is load-bearing: a task.toml with no `repo =` line is perfectly valid
# (single-repo tasks don't need one), but under `set -o pipefail` grep's exit 1
# propagates out of the pipeline and `set -e` would kill the script here.
MEMBER=""
if [ -f "$TASK_DIR/task.toml" ]; then
MEMBER=$(grep -E '^repo[[:space:]]*=' "$TASK_DIR/task.toml" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)
fi
# Single-repo toolkits keep the repo at $ROOT/repo; a polyglot toolkit keeps each member
# at $ROOT/repos/<member>. If the single-repo path is absent, use the member from task.toml
# so a graded task builds against the right member repo.
if [ ! -d "$REPO_DIR/.git" ] && [ -n "$MEMBER" ]; then
if [ -d "$TOOLKIT_ROOT/repos/$MEMBER/.git" ]; then
REPO_DIR="$TOOLKIT_ROOT/repos/$MEMBER"
fi
fi
if [ ! -d "$REPO_DIR/.git" ]; then
echo "Error: repo not found at $REPO_DIR" >&2
exit 1
fi
# Get commit from arg or task.toml
if [ -n "${2:-}" ]; then
COMMIT="$2"
else
# `|| true` for the same reason as MEMBER above: without it, pipefail turns a
# task.toml with no `commit` line into a bare `set -e` abort, and the explicit
# error below never gets a chance to print.
COMMIT=$(grep 'commit' "$TASK_DIR/task.toml" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)
if [ -z "$COMMIT" ]; then
echo "Error: no commit specified and could not read from task.toml" >&2
exit 1
fi
fi
WORKSPACE="$TASK_DIR/environment/workspace"
echo "Building workspace for $TASK_SLUG"
echo " Commit: $COMMIT"
# `browser = true` in task.toml gives the trial Playwright + Chromium. The build has no way
# to read task.toml — a Dockerfile can only see its build context — so the answer is written
# here as a file the Dockerfile COPYs.
#
# ALWAYS write it, including the "0" case: the COPY is unconditional, and a missing source
# fails the build. Accepts `true` and `"true"`, since the quoted form is a plausible hand-edit
# and rejecting it would silently give a task no browser after its author asked for one.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
mkdir -p "$TASK_DIR/environment"
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
# --- DNS jail script staging -------------------------------------------------
# The Dockerfile COPYs this directory, so it must always exist (same rule as the marker
# above: a missing COPY source fails the build). The script itself is optional -- without it
# the image installs no resolver and trials simply run with normal network access.
mkdir -p "$TASK_DIR/environment/dns-jail"
if [ -f "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" ]; then
cp "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" \
"$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
fi
# Not every member's image ships a browser, and the Explore container has one either way — so
# a task can ask for a browser it will not get. Say so here rather than let it pass silently.
if [ "$BROWSER_OPTIN" = "1" ]; then
if grep -q "COPY browser-optin" "$TASK_DIR/environment/Dockerfile" 2>/dev/null; then
echo " Browser: Playwright + Chromium (browser = true)"
else
echo " WARNING: browser = true, but this task's Dockerfile has no browser. The agent" >&2
echo " will get the Read tool and no Chromium. Either drop the flag, or use a" >&2
echo " member whose image ships one:" >&2
echo " grep -l 'COPY browser-optin' task-shared/Dockerfile.*" >&2
fi
fi
# Resolve commit
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
echo " Resolved SHA: $RESOLVED_SHA"
# --- Member-specific setup ---------------------------------------------------
#
# A polyglot _task-scaffold can't know which member a task targets, so anything
# member-specific is resolved here instead of left to the author to remember. This is
# the one step every task runs on both the manual and snapshot paths, and task.toml
# already tells us the member. Both actions below are idempotent and never clobber
# authored content, so re-running is always safe.
SHARED_DIR="$TOOLKIT_ROOT/task-shared"
MEMBER_LC=$(echo "${MEMBER:-}" | tr '[:upper:]' '[:lower:]')
# 1. Base image. Replace the placeholder Dockerfile with the member's real base. Guarded
# on the placeholder marker so an authored Dockerfile is never touched — snapshot tasks
# append session staging to theirs, and any task may be customized by hand. The marker
# must match POLYGLOT_SCAFFOLD_DOCKERFILE in package-worker-toolkit.ts; a packaging test
# asserts the two agree so this can't silently stop matching.
TASK_DOCKERFILE="$TASK_DIR/environment/Dockerfile"
if [ -f "$TASK_DOCKERFILE" ] && grep -q 'POLYGLOT TOOLKIT' "$TASK_DOCKERFILE"; then
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/Dockerfile.$MEMBER_LC" ]; then
cp "$SHARED_DIR/Dockerfile.$MEMBER_LC" "$TASK_DOCKERFILE"
echo " Set base image: environment/Dockerfile (from Dockerfile.$MEMBER_LC)"
else
echo " WARN: environment/Dockerfile is still the scaffold placeholder and no" >&2
echo " task-shared/Dockerfile.${MEMBER_LC:-<member>} exists to replace it with." >&2
echo " Set [metadata].repo in task.toml to your member, then re-run this script." >&2
echo " Members: $(cd "$SHARED_DIR" 2>/dev/null && ls Dockerfile.* 2>/dev/null | sed 's/Dockerfile\.//' | tr '\n' ' ')" >&2
fi
fi
# 2. Deterministic checks (tests/typecheck/lint). tests/test.sh sources this file and
# hands its output to the grader as evidence for the CORRECTNESS score, so a task without
# it gets a correctness score judged from the code alone — no test signal behind it. The
# absent-only guard leaves an existing file untouched (a single-repo scaffold ships one).
if [ ! -f "$TASK_DIR/tests/test-commands.sh" ]; then
CHECKS_SRC=""
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/test-commands.$MEMBER_LC.sh" ]; then
CHECKS_SRC="$SHARED_DIR/test-commands.$MEMBER_LC.sh"
elif [ -f "$SHARED_DIR/test-commands.sh" ]; then
CHECKS_SRC="$SHARED_DIR/test-commands.sh"
fi
if [ -n "$CHECKS_SRC" ]; then
mkdir -p "$TASK_DIR/tests"
cp "$CHECKS_SRC" "$TASK_DIR/tests/test-commands.sh"
chmod +x "$TASK_DIR/tests/test-commands.sh"
echo " Staged deterministic checks: tests/test-commands.sh (from $(basename "$CHECKS_SRC"))"
else
# Say it out loud. Absence is legitimate for members with no runnable checks, but
# silence is indistinguishable from a mistake — and it changes how the correctness
# score is arrived at, so the author should know either way.
echo " NOTE: no deterministic checks available for ${MEMBER:-this repo} — the grader will"
echo " score correctness from the code alone, with no test/typecheck/lint signal."
fi
fi
# Clean and recreate
rm -rf "$WORKSPACE"
mkdir -p "$WORKSPACE"
# Export repo at target commit (no git history).
# --no-same-owner: `git archive` stamps every entry as uid/gid 0, so GNU tar
# running as (container) root tries to chown files back to 0/0. On nested /
# rootless / Sysbox runtimes the container "root" is a userns-mapped uid with no
# CAP_CHOWN, so that chown fails with EPERM. --no-same-owner skips the restore
# (files are owned by the extracting user) — a no-op for real root and for
# non-root extraction, and the fix for the mapped-root case.
git -C "$REPO_DIR" archive "$RESOLVED_SHA" | tar -x --no-same-owner -C "$WORKSPACE"
# Apply workspace patch if one exists
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
if [ -f "$PATCH_FILE" ]; then
echo " Applying workspace.patch..."
cd "$WORKSPACE"
# gc.auto=0 / maintenance.auto=false / gc.autoDetach=false prevent git
# from launching background processes (gc, commit-graph, fsmonitor) that
# can write into .git/objects after the foreground command returns. If
# such a write races with the `rm -rf .git` below, rmdir trips on
# "Directory not empty" and the build fails non-deterministically.
GIT_FLAGS=(-c gc.auto=0 -c gc.autoDetach=false -c maintenance.auto=false)
git "${GIT_FLAGS[@]}" init --quiet
git "${GIT_FLAGS[@]}" add -A
# Inject identity inline so this works on containers without a global
# git config (e.g., native Linux Docker, fresh container images).
# The .git directory is deleted on the next line, so these values are
# throwaway and never reach the patch, the workspace, or the agent.
git "${GIT_FLAGS[@]}" -c user.email=toolkit@local -c user.name=Toolkit commit -m "base" --quiet
git "${GIT_FLAGS[@]}" apply "$PATCH_FILE"
# Belt-and-suspenders: retry rm a few times in case anything still races.
for _ in 1 2 3; do
if rm -rf .git 2>/dev/null; then
break
fi
sleep 0.5
done
# Final attempt without swallowing errors, so a genuine failure surfaces.
if [ -d .git ]; then
rm -rf .git
fi
cd "$TOOLKIT_ROOT"
echo " Patch applied."
fi
# Bundle transitive poetry sibling deps. Some polyglot Python members poetry-depend on
# sibling repos via `ssh://git@github.com/AskZeta/<name>`, which can't resolve in a single-member
# harbor image (no SSH key / network). Archive the transitive closure into workspace/.zeta-siblings/<name>/
# from the toolkit's repos/zeta-<name>/ (members are packaged under their display name zeta-<name>);
# the generated Dockerfile rewrites those git deps to
# these local paths before `poetry install`. No-op for members without such deps.
if [ -f "$WORKSPACE/pyproject.toml" ]; then
SIB_DIR="$WORKSPACE/.zeta-siblings"
queue=("$WORKSPACE/pyproject.toml")
seen=" "
while [ "${#queue[@]}" -gt 0 ]; do
pp="${queue[0]}"; queue=("${queue[@]:1}")
[ -f "$pp" ] || continue
for name in $(grep -oE 'ssh://git@github\.com/AskZeta/[A-Za-z0-9._-]+' "$pp" 2>/dev/null | sed -E 's#.*/AskZeta/##; s#\.git$##' | sort -u); do
case "$seen" in *" $name "*) continue ;; esac
seen="$seen$name "
sib="$TOOLKIT_ROOT/repos/zeta-$name"
[ -e "$sib/.git" ] || { echo " WARN: sibling repo not found: $name" >&2; continue; }
mkdir -p "$SIB_DIR/$name"
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SIB_DIR/$name"
queue+=("$SIB_DIR/$name/pyproject.toml")
done
done
[ -d "$SIB_DIR" ] && echo " Bundled siblings:$(printf '%s' "${seen# }" | sed 's/ $//' | sed 's/^/ /')"
fi
# Bundle Maven sibling libs. The swingbell-polyglot Java services depend on sibling shared
# artifacts (com.swingbell*: common-repository, common-aws-service, jasper-report from
# `reports`, jwt-encryption-decryption) at 0.0.1-SNAPSHOT — resolvable only from a local
# reactor install, never a registry. Walk the dependency closure (a bundled provider's own
# pom can name further siblings — `reports` needs common-repository) into
# workspace/.sbl-siblings/<name>/ with an ORDER file in install order (commons before
# consumers); the generated java Dockerfile `mvn install`s them into ~/.m2 before building
# the member. No-op without a pom or refs.
#
# Twin forks: the book-my-minutes-* repos publish the SAME coordinates as the swingbell
# commons (com.swingbell.common:common-repository:0.0.1-SNAPSHOT etc. — the twin naming is
# repo-level only, invisible to Maven), so an artifactId resolves to the provider from the
# member's own family.
if [ -f "$WORKSPACE/pom.xml" ] && grep -q 'com\.swingbell' "$WORKSPACE/pom.xml" 2>/dev/null; then
SBL_DIR="$WORKSPACE/.sbl-siblings"
case "$MEMBER" in book-my-minutes-*) SBL_TWIN=book-my-minutes- ;; *) SBL_TWIN= ;; esac
queue=("$WORKSPACE/pom.xml")
seen=" "
while [ "${#queue[@]}" -gt 0 ]; do
pom="${queue[0]}"; queue=("${queue[@]:1}")
[ -f "$pom" ] || continue
for artifact in $(grep -oE '<artifactId>(common-repository|common-aws-service|jasper-report|jwt-encryption-decryption)</artifactId>' "$pom" 2>/dev/null | sed -E 's#</?artifactId>##g' | sort -u); do
case "$artifact" in
common-repository|common-aws-service) provider="$SBL_TWIN$artifact" ;;
jasper-report) provider=reports ;;
jwt-encryption-decryption) provider=jwt-encryption-decryption ;;
esac
case "$seen" in *" $provider "*) continue ;; esac
# never bundle the member into itself: the pom's OWN <artifactId> declaration
# matches the grep above just like a dependency would ($MEMBER is the task repo)
[ "$provider" = "$MEMBER" ] && continue
seen="$seen$provider "
sib="$TOOLKIT_ROOT/repos/$provider"
[ -e "$sib/.git" ] || { echo " WARN: maven sibling repo not found: $provider" >&2; continue; }
mkdir -p "$SBL_DIR/$provider"
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SBL_DIR/$provider"
queue+=("$SBL_DIR/$provider/pom.xml")
done
done
# ORDER = canonical install order (providers before their consumers), filtered to the
# closure just bundled — discovery order is consumer-first, which is backwards for install.
for provider in "${SBL_TWIN}common-repository" "${SBL_TWIN}common-aws-service" jwt-encryption-decryption reports; do
case "$seen" in *" $provider "*) echo "$provider" >> "$SBL_DIR/ORDER" ;; esac
done
[ -f "$SBL_DIR/ORDER" ] && echo " Bundled maven siblings: $(tr '\n' ' ' < "$SBL_DIR/ORDER")"
fi
# --- Reference-data corpus: mounted at /data/zeta-corpus in the trial -----------------------
# If this toolkit ships the supplementary data corpus, it's included in every trial — staged into
# the build context + a COPY added to the Dockerfile, so what you see while authoring (bind-mounted
# at /data/zeta-corpus) is exactly what the trial sees. Toolkits without a corpus never include it.
CORPUS_SRC=""
for cand in "${ZETA_CORPUS_DIR:-}" "$TOOLKIT_ROOT/data/zeta-corpus" "/data/zeta-corpus"; do
[ -n "$cand" ] && [ -d "$cand" ] && { CORPUS_SRC="$cand"; break; }
done
if [ -n "$CORPUS_SRC" ]; then
CORPUS_STAGE="$TASK_DIR/environment/corpus"
rm -rf "$CORPUS_STAGE"
# hardlink-stage (cp -al ~free, same filesystem as the toolkit); full copy fallback.
cp -al "$CORPUS_SRC/." "$CORPUS_STAGE" 2>/dev/null || cp -a "$CORPUS_SRC/." "$CORPUS_STAGE"
DF="$TASK_DIR/environment/Dockerfile"
# Wrapped in toolkit-managed sentinels so scripts/check-task-infra.ts can tell
# this append apart from an author's edit — see scripts/lib/task-infra-integrity.ts.
if [ -f "$DF" ] && ! grep -qF 'COPY corpus/ /data/zeta-corpus' "$DF"; then
{ echo ""; echo "# >>> toolkit-managed: corpus >>>"; \
echo "# Reference-data corpus at /data/zeta-corpus (staged by build-workspace)."; \
echo "COPY corpus/ /data/zeta-corpus/"; \
echo "# <<< toolkit-managed <<<"; } >> "$DF"
fi
if [ -f "$TASK_DIR/task.toml" ]; then
CUR=$(grep -oE '^[[:space:]]*storage_mb[[:space:]]*=[[:space:]]*[0-9]+' "$TASK_DIR/task.toml" | grep -oE '[0-9]+' | head -1 || echo 0)
# 10240 = the sandbox disk ceiling (a higher request is rejected downstream).
if [ "${CUR:-0}" -lt 10240 ] && grep -qE '^[[:space:]]*storage_mb[[:space:]]*=' "$TASK_DIR/task.toml"; then
sed -i.bak -E 's/^([[:space:]]*storage_mb[[:space:]]*=[[:space:]]*)[0-9]+/\110240/' "$TASK_DIR/task.toml"
rm -f "$TASK_DIR/task.toml.bak"
fi
fi
echo " Corpus: staged from $CORPUS_SRC -> environment/corpus + Dockerfile COPY (storage_mb>=10240)"
fi
FILE_COUNT=$(find "$WORKSPACE" -type f | wc -l | tr -d ' ')
echo " Workspace: $WORKSPACE ($FILE_COUNT files)"
# Toolkit-managed files. Stamp them if they aren't already (tasks copied from
# _task-scaffold arrive stamped; this covers the ones built by snapshot-to-task), then
# report. Advisory only — this script writes to the Dockerfile itself, so it never
# blocks; harbor-run and submit-task do.
CHECK_INFRA="$TOOLKIT_ROOT/scripts/check-task-infra.ts"
if [ -f "$CHECK_INFRA" ]; then
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" --stamp "$TASK_SLUG") || true
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_SLUG") || true
fi
echo "Done."

View File

@@ -0,0 +1,138 @@
/**
* check-task-infra.ts — report edits to toolkit-managed files.
*
* Called by `scripts/harbor-run` before a trial and by `scripts/submit-task.ts`
* before packaging, so an accidental edit to the trial Dockerfile, the grader
* orchestration, or the grader system prompt surfaces at the moment it matters
* rather than after a submission is reviewed.
*
* Covers two sets: the task's own managed files (environment/Dockerfile,
* tests/test.sh, the grader system prompts) and the toolkit's `scripts/` tree,
* which is checked once per invocation regardless of which task was named.
*
* Always exits 0. Both checks are advisory — see the notes on IntegrityStatus
* and formatIntegrityReport.
*
* Usage:
* npx tsx scripts/check-task-infra.ts <task-slug-or-dir>
* npx tsx scripts/check-task-infra.ts my-task --json
*/
import { existsSync } from 'fs';
import { basename, isAbsolute, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import {
bannerize,
checkTaskInfraIntegrity,
formatIntegrityReport,
writeManagedStamp,
} from './lib/task-infra-integrity.js';
import {
checkToolkitScriptIntegrity,
scriptIntegrityNotice,
} from './lib/toolkit-script-integrity.js';
const argv = yargs(hideBin(process.argv))
.usage('Usage: $0 <task> [options]')
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.option('stamp', {
type: 'boolean',
default: false,
describe:
'Record the managed files as created, so later edits are detectable. No-op if already stamped.',
})
.demandCommand(1, 'Provide a task slug or directory')
.help()
.parseSync();
const log = pino(
{ name: 'check-task-infra', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
const arg = String(argv._[0]);
const toolkitRoot = process.cwd();
// Accept both a bare slug and a path, since harbor-run is invoked with a path
// (`scripts/harbor-run harbor-tasks/<slug>`) and submit-task with a slug.
const taskDir = isAbsolute(arg)
? arg
: existsSync(resolve(toolkitRoot, arg))
? resolve(toolkitRoot, arg)
: join(toolkitRoot, 'harbor-tasks', arg);
if (!existsSync(taskDir)) {
log.fatal({ taskDir }, 'Task directory not found');
process.exit(1);
}
const slug = basename(taskDir);
// --stamp records a task's baseline. Runs at task creation; never overwrites.
if (argv.stamp) {
if (!existsSync(join(toolkitRoot, 'task-shared'))) {
log.debug('Not a worker toolkit (no task-shared/); nothing to stamp');
process.exit(0);
}
const wrote = writeManagedStamp(taskDir, toolkitRoot);
log.debug({ slug, wrote }, wrote ? 'Stamped toolkit-managed files' : 'Already stamped');
process.exit(0);
}
// Toolkit scripts, not this task's files: checked here because this is already the
// preflight both `harbor-run` and `submit-task.ts` reach. An edited script ships no
// trace of itself, only its output — see toolkit-script-integrity.ts.
const scripts = checkToolkitScriptIntegrity(toolkitRoot);
if (scripts.checked) {
const notice = scriptIntegrityNotice(scripts);
if (notice) {
log.warn(
{
edited: scripts.modified.map((f) => f.path),
missing: scripts.missing.map((f) => f.path),
},
'Toolkit scripts need a look'
);
process.stderr.write(`\n${notice}\n\n`);
} else {
log.info({ files: scripts.files.length }, 'Toolkit scripts are unmodified');
}
}
const report = checkTaskInfraIntegrity(taskDir, toolkitRoot);
if (!report.checked) {
log.debug(
'Managed-file check skipped: no task-shared/ here, or the task was authored on a different toolkit generation (its tests/ assets are its own)'
);
process.exit(0);
}
const message = formatIntegrityReport(report);
if (!message) {
log.info({ files: report.files.length }, 'Toolkit-managed files are unmodified');
process.exit(0);
}
// Advisory, always. Exiting non-zero here is what used to let a false positive stop
// an author's trial with no way out; the report is the whole product.
log.warn(
{
edited: report.modified.map((f) => f.taskPath),
outdated: report.outdated.map((f) => f.taskPath),
unverifiable: report.unverifiable.map((f) => f.taskPath),
},
'Toolkit-managed files need a look'
);
process.stderr.write(`\n${bannerize(message, report)}\n\n`);
process.exit(0);

View File

@@ -0,0 +1,291 @@
#!/bin/bash
# Check that a task's live environment/workspace matches what a rebuild from
# the pinned commit + environment/workspace.patch would produce — i.e. the
# workspace every downstream consumer of the task actually sees. Files edited
# (or added/deleted) directly in the built workspace are visible to your local
# trials but do NOT survive packaging: your own tarball may carry them, but
# the finalized task keeps only the rebuild inputs (the workspace/ dir itself
# is gitignored), and everywhere downstream the workspace is rebuilt from the
# gitref in task.toml plus workspace.patch (see build-workspace.sh) — anything
# not captured there is silently dropped.
#
# Usage:
# bash scripts/check-workspace-sync.sh <task-dir> # check (advisory)
# bash scripts/check-workspace-sync.sh --update-patch <task-dir> # fold live edits into workspace.patch
#
# Check mode is run automatically at the start of every `scripts/harbor-run`.
# It warns loudly when the workspace has uncaptured changes, and always exits
# 0 — it never blocks a run. It also exits 0 (silently) when it can't resolve
# the source repo or the pinned commit, since it can't tell anything useful
# then.
#
# --update-patch regenerates environment/workspace.patch as the full diff from
# the pinned commit to the live workspace (the previous patch's changes are
# preserved — they're part of that diff). After updating the patch, re-run
# your trials: reference runs should be captured against the workspace every
# downstream rebuild produces.
#
# Mechanics: the pinned commit's tree is read into a THROWAWAY git index (with
# a throwaway object directory layered over the repo's, so the source repo is
# never written to), workspace.patch is applied to that index, and the live
# workspace directory is compared against it. Files matched by the repo's
# .gitignore are not considered — they can't be captured in workspace.patch
# either, so they never ship either way. File-mode-only changes are ignored
# (core.fileMode=false), matching how patches are generated here.
set -euo pipefail
MODE="check"
if [ "${1:-}" = "--update-patch" ]; then
MODE="update"
shift
fi
if [ -z "${1:-}" ]; then
echo "Usage: $0 [--update-patch] <task-dir>" >&2
exit 1
fi
# Normalize the task dir (tolerates relative paths and trailing slashes).
TASK_DIR="$(cd "$1" 2>/dev/null && pwd)" || {
echo "Error: task directory not found: $1" >&2
exit 1
}
SLUG="$(basename "$TASK_DIR")"
# Tasks live at <root>/harbor-tasks/<slug> in every layout this script ships to.
ROOT="$(cd "$TASK_DIR/../.." && pwd)"
WORKSPACE="$TASK_DIR/environment/workspace"
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
TASK_TOML="$TASK_DIR/task.toml"
# How to spell this script in the recommendations we print. In a packed
# toolkit it lives at <root>/scripts/ (the worker's usual cwd is <root>), so
# the short form works; anywhere else (e.g. invoked from the internal repo
# layout via harbor-run) fall back to the invoked path.
SELF_DISPLAY="bash scripts/check-workspace-sync.sh"
if [ ! -f "$ROOT/scripts/check-workspace-sync.sh" ]; then
SELF_DISPLAY="bash $0"
fi
# In check mode every "can't verify" path exits 0 quietly: this is an advisory
# preflight and a task we can't reason about must never break a run. In
# --update-patch mode the same conditions are hard errors — the user asked for
# a patch and we can't produce one.
skip() {
if [ "$MODE" = "update" ]; then
echo "Error: $1" >&2
exit 1
fi
exit 0
}
[ -d "$WORKSPACE" ] || skip "workspace not built at $WORKSPACE (run build-workspace.sh first)"
[ -f "$TASK_TOML" ] || skip "no task.toml at $TASK_TOML"
# Pinned commit: the `commit = "..."` line in task.toml. Anchored to the line
# start so prose mentions (e.g. a `source = "... commit abc"` note) don't
# match. No commit line is legitimate for some internally-built tasks — then
# there's nothing to compare against.
COMMIT="$(grep -E '^[[:space:]]*commit[[:space:]]*=' "$TASK_TOML" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)"
[ -n "$COMMIT" ] || skip "no commit pinned in task.toml"
# The member this task targets, per task.toml ([metadata].repo) — used to
# resolve the source repo in polyglot layouts. Same extraction as
# build-workspace.sh.
MEMBER="$(grep -E '^repo[[:space:]]*=' "$TASK_TOML" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)"
# Source repo resolution, in order:
# <root>/repo — single-repo toolkit
# <root>/repos/<member> — polyglot toolkit
# <root>/repos/<member>/repo — internal submodule layout
REPO_DIR=""
for cand in "$ROOT/repo" ${MEMBER:+"$ROOT/repos/$MEMBER" "$ROOT/repos/$MEMBER/repo"}; do
if [ -e "$cand/.git" ]; then
REPO_DIR="$cand"
break
fi
done
[ -n "$REPO_DIR" ] || skip "source repo not found under $ROOT"
# Resolve the repo's real git dir (handles submodules, whose .git is a file).
GITDIR="$(git -C "$REPO_DIR" rev-parse --absolute-git-dir 2>/dev/null)" || skip "not a git repo: $REPO_DIR"
RESOLVED_SHA="$(git --git-dir="$GITDIR" rev-parse --quiet --verify "$COMMIT^{commit}" 2>/dev/null)" || \
skip "pinned commit $COMMIT not found in $REPO_DIR"
# --- Throwaway git state ------------------------------------------------------
# A temp index + temp object dir (with the real object dir as a read-only
# alternate) lets us build "commit + patch" as an index and diff the live
# workspace against it without ever writing to the source repo or creating a
# .git inside the workspace.
TMP="$(mktemp -d)"
trap 'rm -rf "$TMP"' EXIT
export GIT_INDEX_FILE="$TMP/index"
export GIT_OBJECT_DIRECTORY="$TMP/objects"
export GIT_ALTERNATE_OBJECT_DIRECTORIES="$GITDIR/objects"
mkdir -p "$GIT_OBJECT_DIRECTORY"
# Suppress mode-bit and line-ending munging so the comparison is about content,
# and keep non-ASCII paths readable instead of C-quoted ("\360\237...").
GIT_FLAGS=(-c core.fileMode=false -c core.autocrlf=false -c core.quotePath=false)
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
# `.zeta-siblings/` is staged INTO the workspace by build-workspace.sh on some
# toolkits (bundled sibling deps) — a build artifact, never part of the patch.
# `.raccoon-setup-done` is run-app's per-repo first-use setup marker (polyglot
# toolkits) — authoring-machine state, never task content. run-app git-ignores
# it via the repo's .git/info/exclude, but this script diffs through a
# throwaway --git-dir that never reads that file, so exclude it here too.
# The leading `.` positive pathspec is load-bearing: several git commands
# reject a pathspec made of nothing but exclusions.
EXCLUDES=("." ":(exclude).zeta-siblings" ":(exclude).raccoon-setup-done")
cd "$WORKSPACE"
export GIT_WORK_TREE="$WORKSPACE"
if [ "$MODE" = "update" ]; then
# Stage the live workspace on top of the pinned tree, then emit the full
# tree -> index diff as the new workspace.patch. --binary --full-index so
# binary additions survive a later `git apply`.
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" add -A -- "${EXCLUDES[@]}"
NEW_PATCH="$TMP/workspace.patch"
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --binary --full-index "$RESOLVED_SHA" -- "${EXCLUDES[@]}" > "$NEW_PATCH"
if [ ! -s "$NEW_PATCH" ]; then
if [ -f "$PATCH_FILE" ]; then
rm -f "$PATCH_FILE"
echo "Workspace matches commit $COMMIT exactly — removed the now-empty environment/workspace.patch."
else
echo "Workspace matches commit $COMMIT exactly — no workspace.patch needed."
fi
exit 0
fi
# Verify the regenerated patch applies to the pristine tree before
# installing it, so we never leave behind a patch build-workspace.sh
# would choke on.
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached --check "$NEW_PATCH" || {
echo "Error: regenerated patch does not apply cleanly to $COMMIT — workspace.patch left unchanged." >&2
exit 1
}
cp "$NEW_PATCH" "$PATCH_FILE"
FILE_COUNT="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --name-only "$RESOLVED_SHA" -- "${EXCLUDES[@]}" | wc -l | tr -d ' ')"
echo "Wrote environment/workspace.patch: $FILE_COUNT file(s) differ from commit $COMMIT."
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
echo ""
echo "NOTE: this task was built from a session snapshot, so workspace.patch is"
echo "meant to mirror the workspace state the captured session describes. Make"
echo "sure these folded-in changes don't contradict the session transcript —"
echo "if they belong to the session's story, re-capturing the snapshot"
echo "(/create-snapshot:snapshot, then scripts/snapshot-to-task.ts) is the"
echo "cleaner fix."
fi
echo ""
echo "Re-run your trials so your reference runs match what now ships:"
echo " scripts/harbor-run harbor-tasks/$SLUG -k 4"
exit 0
fi
# --- Check mode ---------------------------------------------------------------
# Apply workspace.patch to the throwaway index — the index then holds exactly
# the tree build-workspace.sh would produce. A patch that no longer applies is
# its own (serious) problem: the shipped inputs can't even rebuild.
if [ -s "$PATCH_FILE" ]; then
if ! git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached "$PATCH_FILE" 2>/dev/null; then
echo "" >&2
echo "==============================================================================" >&2
echo "!! WARNING: environment/workspace.patch does not apply to commit $COMMIT." >&2
echo "!! A rebuild of this task from its shipped inputs (build-workspace.sh)" >&2
echo "!! would FAIL, and your live workspace can't be checked against them." >&2
echo "!! Did the gitref or the patch change after the workspace was built?" >&2
echo "==============================================================================" >&2
echo "" >&2
exit 0
fi
fi
# Tracked files that differ between the index (commit + patch) and the live
# workspace, plus files that exist only in the live workspace. Both respect
# the repo's .gitignore.
DIFF_RAW="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --name-status -- "${EXCLUDES[@]}")"
UNTRACKED="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" ls-files --others --exclude-standard -- "${EXCLUDES[@]}")"
# Deletions of symlinks are skipped: the internal workspace build prunes
# dangling symlinks after applying the patch, so their absence is expected,
# not a worker edit.
CHANGES=""
while IFS=$'\t' read -r st path; do
[ -n "$st" ] || continue
if [ "$st" = "D" ]; then
entry_mode="$(git --git-dir="$GITDIR" ls-files -s -- "$path" | awk '{print $1}')"
[ "$entry_mode" = "120000" ] && continue
fi
CHANGES="${CHANGES} ${st} ${path}
"
done <<< "$DIFF_RAW"
while IFS= read -r path; do
[ -n "$path" ] || continue
CHANGES="${CHANGES} ?? ${path}
"
done <<< "$UNTRACKED"
[ -n "$CHANGES" ] || exit 0
TOTAL="$(printf '%s' "$CHANGES" | wc -l | tr -d ' ')"
LISTED="$(printf '%s' "$CHANGES" | head -25)"
{
echo ""
echo "=============================================================================="
echo "!! WARNING: environment/workspace has changes that will NOT survive"
echo "!! packaging."
echo "=============================================================================="
echo ""
echo "The workspace/ directory itself is never kept: everywhere downstream the"
echo "task is rebuilt from the commit pinned in task.toml ($COMMIT) plus"
echo "environment/workspace.patch — exactly what scripts/build-workspace.sh"
echo "produces. These $TOTAL file(s) differ from that rebuild, so your local trials"
echo "see them, but they will not survive packaging:"
echo ""
echo "$LISTED"
if [ "$TOTAL" -gt 25 ]; then
echo " ... and $((TOTAL - 25)) more"
fi
echo ""
echo " (M = modified, D = deleted, ?? = only in the live workspace. Files matched"
echo " by the repo's .gitignore are not checked — they never ship either way.)"
echo ""
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
echo "This task was built from a session snapshot, and workspace.patch mirrors"
echo "the workspace state captured with that session. If these changes belong in"
echo "the task, the cleanest fix is to make them in the Explore session and"
echo "re-capture (/create-snapshot:snapshot, then scripts/snapshot-to-task.ts),"
echo "so the session transcript and the workspace stay consistent."
echo ""
echo "To fold them into workspace.patch anyway — only if they don't contradict"
echo "what the captured session says about the workspace:"
echo ""
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
else
echo "To fold these changes into workspace.patch so they ship with the task:"
echo ""
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
if [ -f "$ROOT/scripts/build-workspace.sh" ]; then
echo ""
echo "To discard them instead (rebuild the workspace from commit + patch):"
echo ""
echo " bash scripts/build-workspace.sh $SLUG"
fi
fi
echo ""
echo "Either way, re-run your trials afterwards so your reference runs match the"
echo "workspace every downstream rebuild produces."
echo "=============================================================================="
echo ""
} >&2
exit 0

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,545 @@
"""Custom Codex agents for our devcontainer-based task images.
Harbor's stock Codex agent (`harbor.agents.installed.codex.Codex`) installs Node
via nvm into `$HOME/.nvm` during `install()`, and `setup()` ALWAYS calls
`install()` (the version probe only runs afterward). That install fails on our
task images: they are `FROM mcr.microsoft.com/devcontainers/typescript-node:20`,
which provides Node through the devcontainer nvm at `/usr/local/share/nvm`, and
the tasks run as root, where that nvm isn't auto-loaded — so harbor's
`$HOME/.nvm/nvm.sh` doesn't exist and the agent dies with "NVM failed to load".
`SystemNodeCodex` overrides `install()` to load the image's existing Node and
install only the codex CLI (no second Node via nvm). Everything else — the
trajectory parsing, the codex exec, reasoning_effort kwargs — is inherited
unchanged from the stock agent.
Use via: `--agent-import-path codex_agent:SystemNodeCodex` (PYTHONPATH=scripts).
"""
from __future__ import annotations
import json
import os
import shlex
import sys
import tempfile
import uuid
from pathlib import Path
import atif_session
import browser_note
try:
from dnsjail import apply_dns_jail
except ImportError: # no helper shipped -> no jail, rather than no trials
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
return None
from harbor.agents.installed.codex import Codex
from harbor.models.trial.paths import EnvironmentPaths
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
import codex_auth # noqa: E402
from harness_registry import load_harness_registry # noqa: E402
# Installs the codex CLI into /usr/local/bin so harbor's plain-sh execs find it.
#
# Primary path is the official standalone installer, which fetches a prebuilt
# native binary and needs only curl + tar — no Node in the image. That matters
# because most task images (Ruby/Python) ship no Node at all, and the npm route
# below can only run on the Node-bearing minority.
#
# The npm route is kept as a fallback for images where the installer can't run
# (e.g. a native binary the image's glibc rejects) but a usable npm exists.
_INSTALL_CMD = (
"set -x; "
"if command -v apt-get >/dev/null 2>&1; then "
" apt-get update -qq >/dev/null 2>&1 && apt-get install -y -qq curl ripgrep >/dev/null 2>&1 || true; "
"fi; "
'if ! command -v codex >/dev/null 2>&1; then '
' CODEX_INSTALL_DIR=/usr/local/bin CODEX_NON_INTERACTIVE=true '
' sh -c "curl -fsSL https://chatgpt.com/codex/install.sh | sh" >&2 || true; '
"fi; "
'if ! command -v codex >/dev/null 2>&1 && [ -x "$HOME/.local/bin/codex" ]; then '
' ln -sf "$HOME/.local/bin/codex" /usr/local/bin/codex; '
"fi; "
# npm fallback: load the devcontainer nvm, else find npm anywhere plausible.
'if ! command -v codex >/dev/null 2>&1; then '
' export NVM_DIR="${NVM_DIR:-/usr/local/share/nvm}"; '
' [ -s "$NVM_DIR/nvm.sh" ] && . "$NVM_DIR/nvm.sh" >/dev/null 2>&1 || true; '
' if ! command -v npm >/dev/null 2>&1; then '
' npm_path="$(find /usr/local/share/nvm /usr/local /usr/lib /opt -name npm -type f 2>/dev/null | head -1)"; '
' [ -n "$npm_path" ] && export PATH="$PATH:$(dirname "$npm_path")"; '
" fi; "
' command -v npm >/dev/null 2>&1 && npm install -g @openai/codex@latest; '
"fi; "
'for bin in node codex; do '
' p="$(command -v "$bin" 2>/dev/null || true)"; '
' [ -n "$p" ] && [ "$p" != "/usr/local/bin/$bin" ] && ln -sf "$p" "/usr/local/bin/$bin" || true; '
"done; "
'command -v codex >/dev/null 2>&1 '
' || { echo "FATAL: codex CLI unavailable (standalone installer and npm both failed)" >&2; exit 1; }; '
"codex --version"
)
class SystemNodeCodex(Codex):
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
_has_browser = False
# Non-snapshot codex tasks run this class directly; the native-snapshot resume path
# overrides run() and applies the jail itself.
async def run(self, instruction, environment, context): # type: ignore[override]
self._refuse_shell_hostile_key()
await apply_dns_jail(self, environment)
await super().run(instruction, environment, context)
def _auth_json_setup(self, remote_auth_path: str) -> tuple[dict[str, str], str]:
return codex_auth.auth_json_setup(
self._get_env("OPENAI_API_KEY") or "", remote_auth_path
)
def _refuse_shell_hostile_key(self) -> None:
"""Harbor's own Codex.run interpolates the key into a heredoc, so a key it cannot
escape would 401 with no stated cause. Refuse up front instead."""
if self._resolve_auth_json_path():
return
bad = codex_auth.unescapable_chars(self._get_env("OPENAI_API_KEY") or "")
if bad:
raise ValueError(
"OPENAI_API_KEY contains "
+ ", ".join(repr(c) for c in bad)
+ ", which harbor's stock auth.json writer cannot escape. Point "
"CODEX_AUTH_JSON_PATH at a pre-written auth.json instead."
)
async def install(self, environment) -> None: # type: ignore[override]
await self.exec_as_root(environment, command=_INSTALL_CMD)
self._has_browser = await browser_note.probe_browser(environment)
def build_cli_flags(self) -> str: # type: ignore[override]
"""Harbor's flags plus the registry's `agent_config`, so a trial's toolset
matches the explore launcher's — which passes the same rendering as
$RACCOON_AGENT_FLAGS. Both run paths go through here."""
flags = super().build_cli_flags()
reductions = load_harness_registry().require("codex").agent_config_flags()
if reductions:
flags = f"{flags} {reductions}".strip()
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
def _browser_flag(self) -> str:
"""Disclose the browser to codex the way codex takes extra instructions.
`developer_instructions` PREPENDS a developer message and leaves codex's own base
instructions in place — verified with `codex debug prompt-input`. That makes it the
equivalent of claude's --append-system-prompt. `model_instructions_file`, the other
instruction-shaped key, REPLACES the base instructions; do not use it here.
"""
if not self._has_browser:
return ""
note = browser_note.browser_note()
if not note:
return ""
return f"-c developer_instructions={shlex.quote(note)}"
# Where harbor's run-prep stages the prior Claude Code session for snapshot tasks
# (the same file our snapshot_agent reads). Agent-agnostic, so codex sees it too.
_STAGED_SESSION = "/tmp/snapshot-session/session.jsonl"
def render_claude_session(jsonl_text: str, max_block: int = 4000) -> str:
"""Render a Claude Code session JSONL transcript into readable plain text so a
non-Claude agent (codex) can be handed the prior conversation as context.
Each line is a Claude record: {"type": "user"|"assistant", "message": {"role",
"content"}}. `content` is either a string or a list of blocks
(text / tool_use / tool_result / thinking). We flatten to labeled turns and
truncate oversized tool payloads so the context stays bounded."""
out: list[str] = []
def clip(s: str) -> str:
s = s.rstrip()
return s if len(s) <= max_block else s[:max_block] + "\n…[truncated]"
for line in jsonl_text.splitlines():
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except (json.JSONDecodeError, ValueError):
continue
rtype = rec.get("type")
msg = rec.get("message") or {}
role = msg.get("role") or rtype
content = msg.get("content")
if content is None:
# non-message records (summaries, etc.) — skip unless they carry text
txt = rec.get("summary") or rec.get("content")
if isinstance(txt, str) and txt.strip():
out.append(f"[{rtype}] {clip(txt)}")
continue
if isinstance(content, str):
out.append(f"{role.upper()}: {clip(content)}")
continue
# content is a list of blocks
for block in content:
if not isinstance(block, dict):
out.append(f"{role.upper()}: {clip(str(block))}")
continue
btype = block.get("type")
if btype == "text":
out.append(f"{role.upper()}: {clip(block.get('text', ''))}")
elif btype == "thinking":
out.append(f"{role.upper()} (thinking): {clip(block.get('thinking', ''))}")
elif btype == "tool_use":
name = block.get("name", "?")
inp = json.dumps(block.get("input", {}), ensure_ascii=False)
out.append(f"{role.upper()} [tool_use {name}]: {clip(inp)}")
elif btype == "tool_result":
res = block.get("content")
if isinstance(res, list):
res = "".join(
b.get("text", "") for b in res if isinstance(b, dict)
)
out.append(f"[tool_result]: {clip(str(res))}")
return "\n".join(out)
_INLINE_PREAMBLE = (
"You are continuing an in-progress pair-programming session. Below is the FULL "
"prior conversation between the user and the previous assistant (you), including "
"the tool calls that assistant made and their results. Treat it as your own prior "
"context — the workspace already reflects any edits made in it. Then respond to the "
"user's newest message at the end.\n\n"
"================ PRIOR CONVERSATION ================\n"
)
class InlineSnapshotCodex(SystemNodeCodex):
"""Bridge A: run codex on snapshot tasks by INLINING the prior Claude session as
plain-text context ahead of the user's next-turn instruction. Works for any
provider — codex just sees a long prompt: [rendered prior conversation] + [the
user's newest message]. For non-snapshot tasks (no staged session) it behaves
exactly like the stock codex agent."""
async def run(self, instruction, environment, context): # type: ignore[override]
session_text = ""
try:
result = await environment.exec(
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
)
session_text = (getattr(result, "stdout", "") or "").strip()
except Exception as exc: # best-effort; fall back to bare instruction
self.logger.warning("InlineSnapshotCodex: could not read session: %s", exc)
if session_text:
rendered = render_claude_session(session_text)
if rendered.strip():
instruction = (
_INLINE_PREAMBLE
+ rendered
+ "\n\n================ USER'S NEWEST MESSAGE ================\n"
+ instruction
)
self.logger.info(
"InlineSnapshotCodex: injected %d chars of rendered prior session",
len(rendered),
)
else:
self.logger.warning("InlineSnapshotCodex: session rendered empty")
else:
self.logger.info(
"InlineSnapshotCodex: no staged session (non-snapshot task or empty); "
"running bare instruction"
)
await super().run(instruction, environment, context)
# ---------------------------------------------------------------------------
# Bridge B: native codex resume.
#
# Instead of inlining the whole prior Claude session into one giant prompt
# (Bridge A, which makes codex stall on a ~50k-token blob), we translate the
# staged session into codex's OWN rollout JSONL format, drop it into
# $CODEX_HOME/sessions/<date>/rollout-<ts>-<uuid>.jsonl, and invoke
# `codex exec resume <uuid> -- <instruction>`. codex then treats the prior turns
# as its own conversation history — prompt-cached and incremental — and only has
# to reason about the user's newest message.
#
# We resume by EXPLICIT session id (not --last): --last is cwd-filtered (help:
# "--all ... disables cwd filtering"), and we can't guarantee the rollout's
# recorded cwd matches the sandbox cwd at runtime; an explicit UUID is a direct
# lookup that sidesteps that entirely.
# ---------------------------------------------------------------------------
# A real recorded codex session_meta line (with codex's base_instructions) is the
# most reliable seed for `resume`. The fixture lives in-repo (mounted into the
# devcontainer where this agent code runs); a captured host copy is a secondary
# source, and a synthesized minimal record is the final fallback.
_ROLLOUT_TEMPLATE_CANDIDATES = (
os.path.join(os.path.dirname(os.path.abspath(__file__)), "codex-rollout-template.jsonl"),
"/Users/nickheiner/.claude/jobs/e8fade29/tmp/codex-rollout-template.jsonl",
)
def _now_iso() -> str:
from datetime import datetime, timezone
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z")
def _session_meta(new_id: str, iso_ts: str) -> dict:
"""Return a codex `session_meta` rollout record, reusing the captured real
template (best fidelity for resume) when readable, else a minimal synthesized
one. The id/timestamp are always overwritten with our fresh values."""
for template_path in _ROLLOUT_TEMPLATE_CANDIDATES:
try:
with open(template_path, encoding="utf-8") as fh:
for line in fh:
line = line.strip()
if not line:
continue
rec = json.loads(line)
if rec.get("type") == "session_meta":
rec["timestamp"] = iso_ts
rec.setdefault("payload", {})
rec["payload"]["id"] = new_id
rec["payload"]["timestamp"] = iso_ts
return rec
except (OSError, ValueError):
continue
return {
"timestamp": iso_ts,
"type": "session_meta",
"payload": {
"id": new_id,
"timestamp": iso_ts,
"cwd": "/workspace",
"originator": "codex_exec",
"cli_version": "0.135.0",
"source": "exec",
"thread_source": "user",
"model_provider": "openai",
},
}
def is_codex_rollout(session_jsonl_text: str) -> bool:
"""True when the staged session is already a codex rollout rather than a Claude Code
transcript. Delegates to atif_session, which owns format detection — a second copy of
the record-type set here is how the two would eventually disagree."""
return atif_session.detect_format(session_jsonl_text) == "codex"
def reid_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
"""Re-key an already-native codex rollout onto `new_id` so `codex exec resume
<new_id>` finds it. Conversation records pass through byte-identical — a
codex-authored snapshot resumed by codex needs no translation, which is the
whole fidelity argument for native seeding."""
lines = [json.dumps(_session_meta(new_id, iso_ts))]
for raw in session_jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict) or rec.get("type") == "session_meta":
continue
lines.append(raw)
return lines
def stage_to_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
"""Build a resumable codex rollout from whichever format the snapshot staged."""
if is_codex_rollout(session_jsonl_text):
return reid_codex_rollout(session_jsonl_text, new_id, iso_ts)
return claude_to_codex_rollout(session_jsonl_text, new_id, iso_ts)
def claude_to_codex_rollout(
session_jsonl_text: str, new_id: str, iso_ts: str, max_total: int | None = None
) -> list[str]:
"""Translate a staged Claude Code session into codex rollout JSONL lines so
`codex exec resume` can continue it natively.
Parsing and rendering live in atif_session, which routes every harness pair
through ATIF; this stays as the codex-side entry point. `max_total` is an optional
char cap used by tests; the default is uncapped — our seeded sessions (~20-95k
tokens) fit every supported model's context."""
return atif_session.atif_to_codex_rollout(
atif_session.claude_session_to_atif(session_jsonl_text),
iso_ts,
session_meta=_session_meta(new_id, iso_ts),
max_total=max_total,
)
class NativeSnapshotCodex(SystemNodeCodex):
"""Bridge B: continue the staged Claude session via NATIVE codex resume.
For snapshot tasks we translate `/tmp/snapshot-session/session.jsonl` into a
codex rollout, write it under `$CODEX_HOME/sessions/`, and run
`codex exec resume <uuid> -- <instruction>`. For non-snapshot tasks (no staged
session) we defer to the stock fresh `codex exec` via the base agent."""
async def _read_staged_session(self, environment) -> str:
try:
result = await environment.exec(
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
)
return (getattr(result, "stdout", "") or "").strip()
except Exception as exc: # best-effort
self.logger.warning("NativeSnapshotCodex: could not read session: %s", exc)
return ""
async def run(self, instruction, environment, context): # type: ignore[override]
# NOTE: codex unconditionally declares its `tool_search` (MCP apps tool-
# discovery) tool, which the OpenAI API REJECTS for nano models with HTTP
# 400 "Tool 'tool_search' is not supported". None of codex's knobs
# (--disable tool_search / features.tool_search / enable_mcp_apps=false /
# disabled_tools) suppress it as of codex 0.135, so nano models are NOT
# runnable under this harness. Use a mini (e.g. gpt-5.4-mini) for the small
# end instead. Non-nano models are unaffected.
await apply_dns_jail(self, environment)
session_text = await self._read_staged_session(environment)
if not session_text:
self.logger.info(
"NativeSnapshotCodex: no staged session (non-snapshot task or empty); "
"running stock fresh codex exec"
)
await SystemNodeCodex.run(self, instruction, environment, context)
return
if not self.model_name:
raise ValueError("Model name is required")
model = self.model_name.split("/")[-1]
new_id = str(uuid.uuid4())
iso_ts = _now_iso()
rollout_lines = stage_to_codex_rollout(session_text, new_id, iso_ts)
self.logger.info(
"NativeSnapshotCodex: %s rollout of %d records (~%d chars) for resume %s",
"re-keyed native" if is_codex_rollout(session_text) else "translated Claude",
len(rollout_lines),
sum(len(line) for line in rollout_lines),
new_id,
)
# --- auth/setup: faithful to harbor's Codex.run (OPENAI_API_KEY → auth.json) ---
escaped_instruction = shlex.quote(instruction)
cli_flags = self.build_cli_flags()
cli_flags_arg = (cli_flags + " ") if cli_flags else ""
auth_json_path = self._resolve_auth_json_path()
remote_codex_home = self._REMOTE_CODEX_HOME.as_posix()
remote_secrets_dir = self._REMOTE_CODEX_SECRETS_DIR.as_posix()
remote_auth_path = (self._REMOTE_CODEX_SECRETS_DIR / "auth.json").as_posix()
env: dict[str, str] = {"CODEX_HOME": remote_codex_home}
setup_env: dict[str, str] = {}
await self.exec_as_agent(
environment,
command=(
f'mkdir -p "$CODEX_HOME" {shlex.quote(remote_secrets_dir)} '
f"{shlex.quote(EnvironmentPaths.agent_dir.as_posix())}"
),
env=env,
)
if auth_json_path:
await environment.upload_file(auth_json_path, remote_auth_path)
if environment.default_user is not None:
await self.exec_as_root(
environment,
command=f"chown {environment.default_user} {remote_auth_path}",
)
setup_command = f'ln -sf {shlex.quote(remote_auth_path)} "$CODEX_HOME/auth.json"\n'
else:
env["OPENAI_API_KEY"] = self._get_env("OPENAI_API_KEY") or ""
setup_env, auth_command = self._auth_json_setup(remote_auth_path)
setup_command = (
auth_command
+ f"ln -sf {shlex.quote(remote_auth_path)} \"$CODEX_HOME/auth.json\"\n"
)
if openai_base_url := self._get_env("OPENAI_BASE_URL"):
env["OPENAI_BASE_URL"] = openai_base_url
setup_command += (
'\ncat >>"$CODEX_HOME/config.toml" <<TOML\n'
'openai_base_url = "${OPENAI_BASE_URL}"\n'
"TOML"
)
skills_command = self._build_register_skills_command()
if skills_command:
setup_command += f"\n{skills_command}"
mcp_command = self._build_register_mcp_servers_command()
if mcp_command:
setup_command += f"\n{mcp_command}"
if setup_command.strip():
await self.exec_as_agent(
environment, command=setup_command, env={**env, **setup_env}
)
# --- write the converted rollout into $CODEX_HOME/sessions/<date>/ ---
date_parts = iso_ts[:10].split("-") # YYYY, MM, DD
sessions_dir = f"{remote_codex_home}/sessions/{date_parts[0]}/{date_parts[1]}/{date_parts[2]}"
rollout_name = f"rollout-{iso_ts.replace(':', '-')}-{new_id}.jsonl"
remote_rollout = f"{sessions_dir}/{rollout_name}"
await self.exec_as_agent(
environment, command=f"mkdir -p {shlex.quote(sessions_dir)}", env=env
)
with tempfile.NamedTemporaryFile(
"w", suffix=".jsonl", delete=False, encoding="utf-8"
) as tmp:
tmp.write("\n".join(rollout_lines) + "\n")
host_rollout = tmp.name
try:
await environment.upload_file(host_rollout, remote_rollout)
if environment.default_user is not None:
await self.exec_as_root(
environment,
command=f"chown {environment.default_user} {shlex.quote(remote_rollout)}",
)
finally:
try:
os.unlink(host_rollout)
except OSError:
pass
# --- resume by explicit session id ---
try:
await self.exec_as_agent(
environment,
command=(
"if [ -s ~/.nvm/nvm.sh ]; then . ~/.nvm/nvm.sh; fi; "
f"codex exec resume {new_id} "
"--dangerously-bypass-approvals-and-sandbox "
"--skip-git-repo-check "
f"--model {model} "
"--json "
"--enable unified_exec "
f"{cli_flags_arg}"
"-- "
f"{escaped_instruction} "
f"2>&1 </dev/null | tee {EnvironmentPaths.agent_dir / self._OUTPUT_FILENAME}"
),
env=env,
)
finally:
try:
await self.exec_as_agent(
environment,
command=(
f"mkdir -p {EnvironmentPaths.agent_dir.as_posix()}\n"
'if [ -d "$CODEX_HOME/sessions" ]; then\n'
f" rm -rf {(EnvironmentPaths.agent_dir / 'sessions').as_posix()}\n"
f' cp -R "$CODEX_HOME/sessions" {(EnvironmentPaths.agent_dir / "sessions").as_posix()}\n'
"fi"
),
env=env,
)
except Exception:
pass

View File

@@ -0,0 +1,393 @@
/**
* copy-reference-run.ts - Copy Harbor job trials into a task's reference-runs directory.
*
* Usage:
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
*
* Examples:
* # Copy a single trial
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__3Df3Bjr
*
* # Copy all trials from a job
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__*
*
* What gets copied:
* - verifier/agent-output/ (answer.md, etc.)
* - verifier/reward.txt (the run's reward) + reward-correctness.txt (reads
* N/A by design — correctness lives inside the graded criteria)
* - verifier/reward.json (the machine-readable reward record)
* - verifier/signals-status.txt (whether every deterministic check actually ran,
* i.e. whether the signals the grader was fed are complete)
* - verifier/grade.md + every grade-<N>.md grader sample
* - verifier/grader-result(-<N>).json, grader-stderr(-<N>).log, grader-samples.txt
* - verifier/grader-regime.json (the grading regime this grade actually ran
* under — unrecoverable after the fact, so it must travel with the grade)
* - agent/claude-code.txt or agent/codex.txt (the harness's own log), agent/trajectory.json
* - session.jsonl — the resumable session log, hoisted to the top of the run dir so
* `view-harbor-session.ts <run-dir>/session.jsonl` can load it without further
* indirection. Its location is per harness: Claude Code writes
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
* - config.json, result.json, trial.log
* - input-checksums.json — sha256 checksums of the task inputs the run was
* generated against (prompt, session snapshot, workspace patch, gitref),
* so submit-task.ts can warn when the run goes stale. Copied from the
* trial dir when harbor-run stamped one at launch time (capturedBy:
* 'run' — immune to edits made between the run and this copy);
* otherwise captured here at copy time as a fallback (capturedBy:
* 'copy').
*
* What is NOT copied:
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
* is hoisted out as `session.jsonl` above; everything else here is
* untyped workspace state)
* - artifacts/
*/
import './lib/check-devcontainer';
import {
existsSync,
mkdirSync,
readdirSync,
readFileSync,
rmSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join } from 'path';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyPath, copyTree } from './lib/copy-tree';
import {
captureTaskInputs,
INPUT_CHECKSUMS_FILENAME,
readTaskInputChecksums,
} from './lib/input-checksums';
import { manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
import { readSessionId } from './session-id';
const HARBOR_TASKS_DIR = 'harbor-tasks';
function findTaskDir(trialPrefix: string): string | null {
if (!existsSync(HARBOR_TASKS_DIR)) return null;
const entries = readdirSync(HARBOR_TASKS_DIR, { withFileTypes: true });
for (const entry of entries) {
if (entry.isDirectory() && entry.name.startsWith(trialPrefix)) {
return join(HARBOR_TASKS_DIR, entry.name);
}
}
return null;
}
/** Newest `rollout-*.jsonl` anywhere under a codex `sessions/` tree, or null. */
function newestRollout(sessionsDir: string): string | null {
if (!existsSync(sessionsDir)) return null;
const found: Array<{ path: string; mtime: number }> = [];
const walk = (dir: string) => {
for (const entry of readdirSync(dir, { withFileTypes: true })) {
const full = join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (entry.name.startsWith('rollout-') && entry.name.endsWith('.jsonl')) {
found.push({ path: full, mtime: statSync(full).mtimeMs });
}
}
};
walk(sessionsDir);
if (found.length === 0) return null;
found.sort((a, b) => b.mtime - a.mtime);
return found[0].path;
}
/**
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
* distinct tasks sharing a truncated prefix (e.g. `foo--hash` and `foo--hash-2`,
* both truncating to `foo--ha`) collide: a dirname-prefix scan returns whichever
* sorts first and misroutes the other's trials (observed in the wild as base +
* `-2` reference-runs sharing trial IDs). result.json is written per-trial with
* the real task_name, so it disambiguates exactly. Returns null when result.json
* is absent/unparseable or names a task dir that doesn't exist (caller then falls
* back to the prefix scan).
*/
function findTaskDirByResultJson(trialPath: string): string | null {
const resultPath = join(trialPath, 'result.json');
if (!existsSync(resultPath)) return null;
let taskName: unknown;
try {
taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown }).task_name;
} catch {
return null;
}
if (typeof taskName !== 'string' || taskName.length === 0) return null;
// Hub-published task_names are org-prefixed (`<org>/<slug>`); the dir is bare.
// Inlined (not the shared bareSlug helper) because this script ships in the
// worker toolkit and must not import outside its shipped file set.
const dir = join(HARBOR_TASKS_DIR, taskName.replace(/^[^/]+\//, ''));
return existsSync(dir) ? dir : null;
}
function copyTrial(trialPath: string, destName?: string) {
trialPath = trialPath.replace(/\/$/, '');
if (!existsSync(trialPath)) {
console.error(`Error: ${trialPath} does not exist`);
process.exit(1);
}
// Repair the SOURCE before reading a byte of it. A trial can leave files
// write-only, which locks out their own owner: everything below — reading
// reward.txt, copying agent-output — fails on them, and any that do get
// through land in the task dir, where harbor hashes every file on every
// later trial and one unreadable path aborts the run.
let sourcePerms = null;
try {
sourcePerms = normalizeTreePermissions(trialPath);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${trialPath}: ${String(err)}`);
console.warn(` If the copy below fails on permissions:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
if (sourcePerms && sourcePerms.failures.length > 0) {
console.warn(
`Warning: could not normalize permissions on ${sourcePerms.failures.length} path(s) under ${trialPath}.`
);
console.warn(` If the copy below fails on permissions, run:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
const rewardPath = join(trialPath, 'verifier', 'reward.txt');
if (!existsSync(rewardPath)) {
console.error(`Error: No reward.txt found in ${trialPath}/verifier/`);
process.exit(1);
}
const reward = readFileSync(rewardPath, 'utf-8').trim();
const trialDir = basename(trialPath);
// Trial dir format: <task-slug-truncated>__<trialId>
const separatorIndex = trialDir.lastIndexOf('__');
if (separatorIndex === -1) {
console.error(
`Error: Trial directory '${trialDir}' does not match expected format <slug>__<trialId>`
);
process.exit(1);
}
const trialPrefix = trialDir.substring(0, separatorIndex);
const trialId = trialDir.substring(separatorIndex + 2);
// Prefer the exact task_name from result.json (handles truncated-prefix
// collisions like `foo--hash` vs `foo--hash-2`); fall back to the dirname
// prefix scan only when result.json can't resolve it.
const taskDir = findTaskDirByResultJson(trialPath) ?? findTaskDir(trialPrefix);
if (!taskDir) {
console.error(
`Error: Could not find task directory matching prefix '${trialPrefix}' in ${HARBOR_TASKS_DIR}/`
);
console.error('Available tasks:');
readdirSync(HARBOR_TASKS_DIR).forEach((d) => console.error(` ${d}`));
process.exit(1);
}
// destName (--dest-name) makes a RECORDED rollout id authoritative: the run
// is copied to exactly that name instead of the minted reward-<r>-<id> —
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
// the publish-manifest run_id stay byte-identical by construction.
const dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
if (existsSync(dest)) {
console.warn(`Warning: ${dest} already exists, overwriting`);
rmSync(dest, { recursive: true });
}
mkdirSync(dest, { recursive: true });
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
// companion reward-correctness.txt (N/A by design) and the machine-readable
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
// (the source of truth grade.md/reward.txt are rendered from), the
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
// render-stderr(-<N>).log, and grader-samples.txt.
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
// every per-sample record.)
//
// signals-status.txt qualifies the deterministic signals the grader was fed:
// it records whether every deterministic check actually produced a verdict
// ("ok") or one or more was killed before finishing ("degraded" — the grade
// is then NOT fully signal-backed). Without it a copied run is
// indistinguishable from a run whose checks all passed, so it must travel
// with the reward files.
const verifierDir = join(trialPath, 'verifier');
if (existsSync(verifierDir)) {
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
if (!entry.isFile()) continue;
const f = entry.name;
if (
f === 'reward.txt' ||
f === 'reward-correctness.txt' ||
f === 'reward.json' ||
f === 'signals-status.txt' ||
f === 'grader-samples.txt' ||
f === 'grader-regime.json' ||
/^grade(-\d+)?\.md$/.test(f) ||
/^grade(-\d+)?\.json$/.test(f) ||
/^grader-result(-\d+)?\.json$/.test(f) ||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
/^render-stderr(-\d+)?\.log$/.test(f)
) {
copyPath(join(verifierDir, f), join(dest, f));
}
}
}
const agentOutputDir = join(verifierDir, 'agent-output');
if (existsSync(agentOutputDir)) {
copyTree(agentOutputDir, join(dest, 'agent-output'));
}
// Copy agent session log and trajectory (not workspace)
const agentDir = join(trialPath, 'agent');
if (existsSync(agentDir)) {
mkdirSync(join(dest, 'agent'), { recursive: true });
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
// the Claude one left codex reference runs with nothing but the trajectory.
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
const src = join(agentDir, file);
if (existsSync(src)) copyPath(src, join(dest, 'agent', file));
}
// The grader (and downstream worldbench export / replay) reads
// agent/trajectory.json. If it's missing, scream so we don't silently
// ship a reference run that's only half-useful.
if (!existsSync(join(agentDir, 'trajectory.json'))) {
console.warn(
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
`Harbor's adapter for this harness failed to write it (typically because ` +
`the converter choked on the session log). Downstream consumers ` +
`(grader replay, worldbench export) need this file — investigate ` +
`before relying on this reference run.`
);
}
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
// gets embedded in claude-code.txt's first non-empty line; we use it to
// locate the sibling JSONL Claude Code wrote in the same trial.
// Layout is per harness, so each needs a case here — the same reason
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
// session, which is what codex got before this: nothing at all.
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
if (existsSync(claudeCodeTxt)) {
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
const sessionId = readSessionId(claudeCodeTxt);
if (sessionId) {
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
if (existsSync(jsonl)) copyPath(jsonl, join(dest, 'session.jsonl'));
}
} else {
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
const rollout = newestRollout(join(agentDir, 'sessions'));
if (rollout) copyPath(rollout, join(dest, 'session.jsonl'));
}
}
// Copy top-level metadata
for (const file of ['config.json', 'result.json', 'trial.log']) {
const src = join(trialPath, file);
if (existsSync(src)) copyPath(src, join(dest, file));
}
// Record the checksums of the task inputs this run was generated against
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
// submit-task.ts re-captures at packaging time and warns when any of them
// changed — the run then describes an older revision of the task than the
// one being shipped. Preferred source: the launch-time stamp harbor-run
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
// 'run') — it records the inputs the agent actually ran against, so an
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
// (trials from an older harbor-run, or a failed stamp): capture here at
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
// the weaker evidence.
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
const trialStamp = readTaskInputChecksums(trialStampPath);
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
// harbor-runs sharing a cwd) — its hashes describe some other task's
// inputs, so treat it as absent rather than importing false evidence.
const stampSlug = trialStamp?.taskSlug;
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
if (trialStamp && !stampMisrouted) {
copyPath(trialStampPath, join(dest, INPUT_CHECKSUMS_FILENAME));
} else {
if (stampMisrouted) {
console.warn(
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
);
}
writeFileSync(
join(dest, INPUT_CHECKSUMS_FILENAME),
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
);
}
// Files captured from a run can land unreadable to you, which makes packaging
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
let perms = null;
try {
perms = normalizeTreePermissions(dest);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
console.warn(` The run copied fine. If packaging later fails on permissions:`);
console.warn(` ${manualRepairHint(dest)}`);
}
if (perms && perms.failures.length > 0) {
console.warn(
`Warning: could not normalize permissions on ${perms.failures.length} path(s) under ${dest}.`
);
console.warn(
` If packaging later fails with 'Cannot stat: Permission denied', run:\n` +
` ${manualRepairHint(dest)}`
);
}
console.log(`Copied to ${dest}`);
console.log(` reward: ${reward}`);
console.log(` task: ${taskDir}`);
console.log(` trial: ${trialId}`);
const ownerFixed = (sourcePerms?.ownerFixed.length ?? 0) + (perms?.ownerFixed.length ?? 0);
const modeFixed = (sourcePerms?.modeFixed.length ?? 0) + (perms?.modeFixed.length ?? 0);
if (ownerFixed > 0 || modeFixed > 0) {
console.log(` perms: normalized ${ownerFixed} owner / ${modeFixed} mode`);
}
}
// Main
const rawArgs = process.argv.slice(2);
let destName: string | undefined;
const args: string[] = [];
for (let i = 0; i < rawArgs.length; i++) {
if (rawArgs[i] === '--dest-name') {
destName = rawArgs[++i];
if (!destName) {
console.error('Error: --dest-name requires a value');
process.exit(1);
}
} else {
args.push(rawArgs[i]);
}
}
if (args.length === 0) {
console.error(
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]'
);
process.exit(1);
}
if (destName && args.length !== 1) {
console.error('Error: --dest-name applies to exactly one trial path');
process.exit(1);
}
for (const trialPath of args) {
copyTrial(trialPath, destName);
}

View File

@@ -0,0 +1,98 @@
"""Apply the DNS jail to a trial container: the model endpoint resolves, nothing else does.
Opt-in with RACCOON_DNS_JAIL=1. Runs from the agent's own turn rather than from a compose
overlay — the allowlist comes from the proxy URL this process already holds (plus any hosts
RACCOON_DNS_JAIL_ALLOW adds), so nothing has to be injected into the container, and the jail works on every harbor backend. Deliberately
after agent-setup: a harness that downloads its CLI there still reaches the network to do it.
"""
import logging
import os
import shlex
from typing import Any
JAIL = "/usr/local/bin/raccoon-dns-jail"
_NO_SCRIPT = "raccoon-dns-jail: not in this image"
_URL_VARS = (
"ANTHROPIC_BASE_URL", "OPENAI_BASE_URL", "GOOGLE_GEMINI_BASE_URL",
"HTTPS_PROXY", "https_proxy", "HTTP_PROXY", "http_proxy", "ALL_PROXY", "all_proxy",
)
_EXTRA_VAR = "RACCOON_DNS_JAIL_ALLOW"
_log = logging.getLogger(__name__)
def _host(url: str) -> str:
"""Hostname out of a URL, or "" when it is not a plain hostname we can allow."""
h = url.split("://", 1)[-1].split("/", 1)[0].rsplit("@", 1)[-1].split(":", 1)[0]
if not h or h.startswith((".", "-")) or h.endswith(".") or not all(
c.isascii() and (c.isalnum() or c in ".-") for c in h
):
return ""
# An IP-literal endpoint (a loopback proxy shim, say) needs no DNS at all, and a
# --server rule for it would only be checked by a PTR query the catch-all answers.
if all(part.isdigit() for part in h.split(".")):
return ""
return h
def dns_jail_allowlist() -> tuple[list[str], list[str]]:
"""(required, advisory).
Required = the hosts this process's own env says the agent will dial; every one must
resolve through the jail or no jail is applied, because a host the agent needs and
cannot resolve is a dead trial. Advisory = whatever RACCOON_DNS_JAIL_ALLOW adds, which
only warns: an added host that CNAMEs outside the allowlist cannot resolve through the
catch-all, and must not take the whole jail down with it.
"""
required: list[str] = []
for var in _URL_VARS:
h = _host(os.environ.get(var) or "")
if h and h not in required:
required.append(h)
advisory: list[str] = []
for entry in (os.environ.get(_EXTRA_VAR) or "").replace(",", " ").split():
# Bare hostnames only: a URL silently truncated to its first path segment would
# allow a name nobody asked for and block the one they meant.
h = "" if ("/" in entry or ":" in entry) else _host(entry)
if not h:
_log.warning("DNS jail: ignoring unusable %s entry %r", _EXTRA_VAR, entry)
elif h not in required and h not in advisory:
advisory.append(h)
return required, advisory
def dns_jail_enabled() -> bool:
return os.environ.get("RACCOON_DNS_JAIL") == "1"
async def apply_dns_jail(agent: Any, environment: Any) -> None:
"""No-op unless enabled; leaves the container's DNS untouched on any doubt."""
if not dns_jail_enabled():
return
required, advisory = dns_jail_allowlist()
allow = " ".join(required)
# A blank allowlist means no model endpoint was found: jailing would strand the agent.
if not allow:
_log.warning("DNS jail: no usable model endpoint — the trial keeps normal network access")
return
try:
result = await agent.exec_as_root(
environment,
command=(
f"if [ -x {JAIL} ]; then DNSJAIL_ALLOW={shlex.quote(allow)} "
f"DNSJAIL_ALLOW_EXTRA={shlex.quote(' '.join(advisory))} {JAIL}; "
f'else echo "{_NO_SCRIPT}"; fi'
),
)
except Exception as exc: # a jail that cannot be applied must not fail the trial
_log.warning("DNS jail: could not apply (%s) — the trial keeps normal network access", exc)
return
# An image frozen before this feature has nothing to invoke. Say so: a launcher that
# believes the network is restricted when it is not is worse than no jail at all.
if _NO_SCRIPT in (getattr(result, "stdout", "") or ""):
_log.warning(
"DNS jail: this task's image ships no resolver — the trial keeps normal network access"
)

View File

@@ -0,0 +1,48 @@
#!/bin/bash
# guidance-target.sh — print the holistic-rubric file the grader reads for a task.
#
# The grader reads the task's holistic rubric under the Grading Standard.
# Detector skills call this resolver so they always assess the file the grader
# will actually read, and so the resolution rule lives in one place.
#
# Resolution order (renames are forward-only, so every generation stays readable):
# tests/holistic-rubric.md the current name; new tasks use it
# tests/grader-guidance-consolidated.md tasks created before the rename
# tests/grader-guidance.md legacy-generation tasks
# When none exists yet, the current name is printed — that is the file a new
# task's rubric will be written to.
#
# Usage:
# bash scripts/guidance-target.sh <slug-or-task-dir>
#
# Output (one line): the path to the rubric file.
set -eu
arg="${1:?usage: bash scripts/guidance-target.sh <slug-or-task-dir>}"
dir="$arg"
[ -d "$dir" ] || dir="harbor-tasks/$arg"
tests="$dir/tests"
[ -d "$tests" ] || { echo "ERROR: no tests/ directory at $dir" >&2; exit 1; }
new="$tests/holistic-rubric.md"
old="$tests/grader-guidance-consolidated.md"
legacy="$tests/grader-guidance.md"
if [ -f "$new" ] && [ -f "$old" ]; then
# Both names present: the grader's pick depends on the harness generation,
# so an assessment of either could be an assessment of the wrong file.
# Byte-identical copies are safe; anything else is a hard stop.
if ! cmp -s "$new" "$old"; then
echo "ERROR: $tests carries both holistic-rubric.md and grader-guidance-consolidated.md with different content — keep exactly one (tests/holistic-rubric.md is the current name)" >&2
exit 1
fi
echo "$new"
elif [ -f "$new" ]; then
echo "$new"
elif [ -f "$old" ]; then
echo "$old"
elif [ -f "$legacy" ]; then
echo "$legacy"
else
echo "$new"
fi

View File

@@ -0,0 +1,299 @@
#!/bin/bash
# Re-grade an existing reference run without re-invoking the agent.
#
# Spins up a normal harbor trial, but plugs in scripts/replay_agent.py
# instead of a real agent. The replay agent overlays the captured
# agent-output into /workspace, applies any captured deletions, drops
# the captured trajectory at /logs/agent/trajectory.json so the grader
# reads the same transcript it would for the original run, then exits.
# The verifier (real test.sh, real LLM grader if present) runs as it
# would for any other trial.
#
# Usage:
# scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
#
# Examples:
# # Single regrade
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
#
# # Ten regrades of the same reference run (independent grader trials)
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg \
# -k 10
#
# --fast runs the GRADER in claude's fast serving mode (faster output at a higher
# token rate). The replay agent runs no model, so the grader is the only model in
# this path. Serving speed and cost change; the grade itself is not steered.
#
# See scripts/replay_agent.py for what the agent actually does, and the
# `verifier: capture tracked-file deletions in agent-output` PR for the
# capture half of this flow (_HARBOR_DELETIONS.txt).
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# Source API key + any verifier env from the repo's .env
if [ -f "$REPO_ROOT/.env" ]; then
set -a
source "$REPO_ROOT/.env"
set +a
fi
# The grader authenticates with ANTHROPIC_API_KEY straight out of the .env sourced above, so
# a .env saved on Windows would hand it a value with a carriage return still attached.
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
harness_setup_credentials >/dev/null 2>&1 || true
fi
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi
# When running inside a devcontainer, harbor needs HOST paths for docker
# bind mounts (the docker daemon is on the host).
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
cd "$HOST_WORKSPACE"
fi
usage() {
cat >&2 <<EOF
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
Required arguments:
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
agent-output/ (and ideally agent/trajectory.json).
Optional arguments:
--fast grade in claude's fast serving mode (higher token rate,
faster output). Anything else is passed through to harbor.
EOF
exit 1
}
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
FAST_REQUESTED=""
REGRADE_ARGS=()
while [ $# -gt 0 ]; do
case "$1" in
--fast) FAST_REQUESTED=1; shift ;;
*) REGRADE_ARGS+=("$1"); shift ;;
esac
done
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
[ $# -lt 2 ] && usage
TASK_DIR="$1"
REF_RUN_DIR="$2"
shift 2
# Resolve to absolute paths — harbor cd's around internally; the replay
# agent receives the path as an --agent-kwarg and won't know our cwd.
TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
echo "Error: task-dir does not exist: $TASK_DIR" >&2
exit 1
}
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
exit 1
}
# Newer-generation Dockerfiles COPY environment/dns-jail/, a derived directory
# that harbor-run pre-stages but a bare regrade context may lack — the sandbox
# build then fails before the verifier ever starts. Recreate it the same way.
if [ -d "$TASK_DIR_ABS/environment" ]; then
mkdir -p "$TASK_DIR_ABS/environment/dns-jail"
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
break
fi
done
fi
# Rubric grader modes (HARBOR_GRADER_MODE=rubric-*) read tests/render-rubric-grade.py,
# a shared asset like the dns-jail script above. A task created before the rubric
# renderer shipped has no copy, and a stale copy aggregates with outdated weights;
# either way the regrade must run against the current shared copy. Stage it
# host-side — the container only ever sees the task directory. The source lives at
# harbor-tasks/raccoon-shared/ in the internal repo and task-shared/ in a worker
# toolkit checkout; first one present wins.
case "${HARBOR_GRADER_MODE:-}" in
rubric-*)
RUBRIC_RENDER_DEST="$TASK_DIR_ABS/tests/render-rubric-grade.py"
for RUBRIC_RENDER_SRC in "$REPO_ROOT/harbor-tasks/raccoon-shared/render-rubric-grade.py" \
"$REPO_ROOT/task-shared/render-rubric-grade.py"; do
[ -f "$RUBRIC_RENDER_SRC" ] || continue
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
mkdir -p "$TASK_DIR_ABS/tests"
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
fi
break
done
;;
esac
# NOTE: agent-output/ is intentionally NOT required here. Advisory tasks (the
# agent only reads + answers in chat) make no workspace edits, so a faithful
# capture has an empty/absent agent-output/ — the deliverable lives in the
# captured transcript (agent/trajectory.json) that the grader reads. ReplayAgent
# overlays agent-output/ when present and otherwise grades base-workspace +
# transcript, but FAILS LOUDLY if the transcript shows file-mutating tool calls
# with no agent-output/ (genuine lost edits). So we let it make that call.
if [ ! -d "$REF_RUN_DIR_ABS/agent-output" ]; then
echo "Note: $REF_RUN_DIR_ABS has no agent-output/ — replaying as an" >&2
echo " advisory run (base workspace + captured transcript). See" >&2
echo " scripts/replay_agent.py for the lost-edits safety guard." >&2
fi
# Make scripts/ importable so harbor can find replay_agent:ReplayAgent.
# ${PYTHONPATH:+...} so an unset PYTHONPATH doesn't leave a trailing colon —
# python treats the resulting empty entry as the CWD, silently putting
# whatever directory the user ran this from on harbor's sys.path.
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
# Environment backend. Explicit HARBOR_ENV wins; otherwise default to docker in
# a worker-toolkit checkout (detected by toolkit.json at the repo root) and
# daytona in the internal repo. See scripts/harbor-run for the full rationale
# (why the toolkit needs docker, why the marker is a workspace file not an image
# env, and why daytona must NOT pass --no-delete — billed sandbox).
if [ -n "${HARBOR_ENV:-}" ]; then
ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
ENV_TYPE="docker"
else
ENV_TYPE="daytona"
fi
DELETE_FLAGS="--no-delete"
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
RESOURCE_FLAGS=""
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
exit 1
fi
# Output dir. harbor names the job subdir by second-granularity timestamp, so
# many regrades launched in the same second under one -o collide
# ("Job directory ... already exists and cannot be resumed"). Set
# HARBOR_REGRADE_OUT to a per-run unique dir when running a parallel sweep.
OUT_DIR="${HARBOR_REGRADE_OUT:-harbor-jobs}"
# Optional grader mode: HARBOR_GRADER_MODE=one-shot flips the task's test.sh into
# the no-tools one-shot grader (vs the default agentic grader) via verifier env —
# lets us A/B the agenticity gap without forking the task. See raccoon-shared/test.sh.
GRADER_MODE_FLAG=()
[ -n "${HARBOR_GRADER_MODE:-}" ] && GRADER_MODE_FLAG=(--verifier-env "GRADER_MODE=$HARBOR_GRADER_MODE")
# Optional grader model: HARBOR_GRADER_MODEL=claude-fable-5-1 overrides the grader's
# model (default: the `opus` alias) via verifier env — lets us A/B the grader model
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
# For measuring per-sample properties of the grader (e.g. how often it emits a
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
# env var Claude Code itself reads, so test.sh needs no knowledge of it). For
# proxies whose responses outlast the CLI default.
[ -n "${HARBOR_API_TIMEOUT_MS:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "API_TIMEOUT_MS=$HARBOR_API_TIMEOUT_MS")
# Fast serving mode for the grader's own claude calls (--fast). Per-task tests/ assets
# are frozen at creation, so say when this task's copy cannot act on the flag. The note
# names that copy, not a shared path — this script ships to workers under another layout.
if [ -n "$FAST_REQUESTED" ]; then
GRADER_MODE_FLAG+=(--verifier-env "GRADER_FAST_MODE=true")
if [ -f "$TASK_DIR_ABS/tests/test.sh" ] &&
! grep -q 'GRADER_FAST_MODE' "$TASK_DIR_ABS/tests/test.sh"; then
echo "Note: --fast passed, but this task's tests/test.sh does not read" >&2
echo " GRADER_FAST_MODE, so the grader will run at normal speed —" >&2
echo " its copy predates the flag. Refresh the task's tests/test.sh" >&2
echo " from the current shared grader assets to enable it." >&2
fi
fi
# Carry the SOURCE run's agent identity + model into this replay's own record.
#
# A replay reports `replay_agent:ReplayAgent` with model_name null, because no model
# ran — the behaviour being graded came from the source run. Recording only
# reference_run_dir makes that a pointer, and pointers dangle: a regrade is normally
# copied back over the run it regraded, so the source usually no longer exists (501 of
# 643 on-disk replays already point at a missing dir, none of them in the published
# manifest either). Stamping the values here makes the replay self-describing, so the
# originating harness and model survive the source's deletion.
#
# Regrading a REGRADE means the source is itself a replay, so copying its own identity
# forward would overwrite the real provenance with a self-reference: inherit what it
# inherited instead.
#
# Read with python3 rather than jq — jq is not guaranteed on a worker's box, and a
# missing source result.json must degrade to "unknown", never abort the regrade.
SOURCE_PROV_FLAGS=()
if [ -f "$REF_RUN_DIR_ABS/result.json" ]; then
SOURCE_PROV=$(python3 -c '
import json, sys
try:
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
except Exception:
sys.exit(0)
kw = a.get("kwargs") or {}
REPLAY = "replay_agent:ReplayAgent"
if (a.get("import_path") or a.get("name")) == REPLAY:
agent, model = kw.get("source_agent_import_path"), kw.get("source_model_name")
else:
agent, model = a.get("import_path") or a.get("name"), a.get("model_name")
# An already-damaged chain cannot be recovered; leave it honestly unstamped rather
# than propagating a source that names the replay agent itself.
print("" if agent == REPLAY else agent or "")
print(model or "")
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
SOURCE_AGENT=$(printf '%s\n' "$SOURCE_PROV" | sed -n 1p)
SOURCE_MODEL=$(printf '%s\n' "$SOURCE_PROV" | sed -n 2p)
[ -n "$SOURCE_AGENT" ] && SOURCE_PROV_FLAGS+=(--ak "source_agent_import_path=$SOURCE_AGENT")
[ -n "$SOURCE_MODEL" ] && SOURCE_PROV_FLAGS+=(--ak "source_model_name=$SOURCE_MODEL")
fi
# HARBOR_GRADING_STANDARD, when set, is passed through to the task's own
# tests/test.sh as the GRADING_STANDARD verifier env var. What (if anything)
# it does is decided by the scripts inside that tests/ directory; a test.sh
# that reads no such variable ignores it.
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
# nothing to restrict — only the verifier runs, and it needs the proxy.
exec harbor run \
-p "$TASK_DIR_ABS" \
--agent-import-path replay_agent:ReplayAgent \
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
${SOURCE_PROV_FLAGS[@]+"${SOURCE_PROV_FLAGS[@]}"} \
-e "$ENV_TYPE" \
$DELETE_FLAGS \
$RESOURCE_FLAGS \
--yes \
-o "$OUT_DIR" \
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
"$@"

View File

@@ -0,0 +1,467 @@
#!/bin/bash
# Run raccoon tasks via Harbor with standard defaults.
#
# Automatically detects snapshot-based tasks (those with environment/session.jsonl)
# and uses the snapshot agent adapter for session resume.
#
# Usage: scripts/harbor-run <task-dir> [extra harbor args...]
# Example: scripts/harbor-run harbor-tasks/my-task-slug
# Example: scripts/harbor-run harbor-tasks/my-task-slug -k 4 --force-build
#
# To change the model, use --model (consumed here). Passing harbor's own -m does NOT
# override: harbor's -m is repeatable and builds one agent per value, so `-m X` runs the
# registry default AND X — two trials.
#
# --fast runs both models of the trial in fast serving mode (faster output at a higher
# token rate): the trial agent (claude-code only) and the grader the verifier launches.
# Serving speed and cost change; the grade itself is not steered.
#
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# An explicit RACCOON_DNS_JAIL from the caller wins over .env, so a worker who opted in
# there can still turn the jail off for one run.
_RJ_SET="${RACCOON_DNS_JAIL+set}"; _RJ_VAL="${RACCOON_DNS_JAIL:-}"
_RJA_SET="${RACCOON_DNS_JAIL_ALLOW+set}"; _RJA_VAL="${RACCOON_DNS_JAIL_ALLOW:-}"
# Source API key
if [ -f "$REPO_ROOT/.env" ]; then
set -a
source "$REPO_ROOT/.env"
set +a
fi
if [ -n "$_RJ_SET" ]; then RACCOON_DNS_JAIL="$_RJ_VAL"; fi
if [ -n "$_RJA_SET" ]; then RACCOON_DNS_JAIL_ALLOW="$_RJA_VAL"; fi
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi
# Per-harness credentials (OPENAI_API_KEY and friends) are DERIVED from the proxy root in
# ANTHROPIC_BASE_URL — they are not in .env. The container-create derivation exported them
# into a process that has long since exited, and only the auth FILES it wrote survive, so a
# fresh shell has the key on disk but not in its environment. resolve_harness checks the
# environment, and the trial passes it through to the sandbox, so re-derive here.
# Quiet on purpose: if it does not work, resolve_harness refuses by name a second later.
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
harness_setup_credentials >/dev/null 2>&1 || true
fi
export ANTHROPIC_BASE_URL="${ANTHROPIC_BASE_URL:-}"
# When running inside a devcontainer, harbor computes absolute paths for
# Docker bind mounts. These paths must be HOST paths because docker compose
# talks to the host daemon via the shared socket. Switching CWD to the
# host-equivalent workspace path makes harbor resolve paths correctly.
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
cd "$HOST_WORKSPACE"
fi
TASK_DIR="$1"
shift
# Harness selection. `--harness` / `--model` / `--fast` / `--check-model` are consumed
# here; everything else passes through to harbor untouched, so existing invocations keep
# working. Parsed with a loop rather than getopts because the remaining args are an
# opaque harbor passthrough that getopts would try to interpret.
HARNESS_ARGS=()
PASSTHROUGH=()
FAST_REQUESTED=""
while [ $# -gt 0 ]; do
case "$1" in
--harness) HARNESS_ARGS+=(--harness "$2"); shift 2 ;;
--harness=*) HARNESS_ARGS+=(--harness "${1#*=}"); shift ;;
--model) HARNESS_ARGS+=(--model "$2"); shift 2 ;;
--model=*) HARNESS_ARGS+=(--model "${1#*=}"); shift ;;
--fast) HARNESS_ARGS+=(--fast); FAST_REQUESTED=1; shift ;;
--check-model) HARNESS_ARGS+=(--check-model); shift ;;
*) PASSTHROUGH+=("$1"); shift ;;
esac
done
set -- "${PASSTHROUGH[@]+"${PASSTHROUGH[@]}"}"
# Preflight: workspace must be populated before harbor tries to docker-build it.
# Without this, the Dockerfile's `COPY workspace/ .` fails with an opaque
# "failed to calculate checksum of ref ...: \"/workspace\": not found" buried
# several frames deep in harbor's asyncio + docker-compose traceback. Surface
# the real fix here instead.
WORKSPACE_DIR="$TASK_DIR/environment/workspace"
if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ]; then
SLUG="$(basename "$TASK_DIR")"
echo "Error: $WORKSPACE_DIR is missing or empty." >&2
echo "Build it first: bash scripts/build-workspace.sh $SLUG" >&2
echo "(reads commit from $TASK_DIR/task.toml; applies environment/workspace.patch if present.)" >&2
exit 1
fi
# Preflight: refuse a task dir harbor would not accept as a task.
#
# `harbor run -p` falls back to reading a rejected dir as a DATASET of tasks, finds none,
# and dies with "Either datasets or tasks must be provided." — naming neither the path nor
# the missing file. Name it here instead. Harbor's own interpreter (the uv-tool venv, per
# the shim's shebang) is the only one that can import harbor; skip the check when it or the
# helper is absent, so a stripped environment never blocks a runnable task.
VALIDATE_TASK_DIR="$SCRIPT_DIR/validate_task_dir.py"
# Harbor's own interpreter is the only one that can import harbor. RACCOON_HARBOR_PYTHON
# overrides it for tests, which have no harbor to read a shebang from.
HARBOR_PY="${RACCOON_HARBOR_PYTHON:-}"
if [ -z "$HARBOR_PY" ] && HARBOR_BIN="$(command -v harbor 2>/dev/null)"; then
HARBOR_PY="$(sed -n '1s|^#!||p' "$HARBOR_BIN" 2>/dev/null || true)"
fi
if [ -f "$VALIDATE_TASK_DIR" ] && [ -n "$HARBOR_PY" ] && [ -x "${HARBOR_PY%% *}" ]; then
# --install-only implies --disable-verification in harbor, for task validation too.
VALIDATE_ARGS=()
case " $* " in
*" --disable-verification "* | *" --install-only "*) VALIDATE_ARGS+=(--disable-verification) ;;
esac
VALIDATE_ERR="$(mktemp)"
# Gate on the printed verdict, never the exit code: an interpreter that cannot run
# the helper at all (a stub harbor with a bash shebang) also exits non-zero, and a
# preflight that can refuse a run must never refuse one that would have worked.
VALIDATE_OUT="$("$HARBOR_PY" "$VALIDATE_TASK_DIR" "$TASK_DIR" \
${VALIDATE_ARGS[@]+"${VALIDATE_ARGS[@]}"} 2>"$VALIDATE_ERR")" || true
if [ "$VALIDATE_OUT" = "verdict=invalid" ]; then
echo "Error: harbor will not accept $TASK_DIR as a task." >&2
sed 's|^| |' "$VALIDATE_ERR" >&2
echo "Fix the file named above, then re-run; harbor's own error names no file." >&2
if [ -d "$REPO_ROOT/harbor-tasks/_task-scaffold" ]; then
echo "A missing toolkit-managed file (tests/test.sh, tests/render-grade-consolidated.py)" >&2
echo "can be copied from harbor-tasks/_task-scaffold/ at the same relative path. Never" >&2
echo "overwrite a task.toml or instruction.md you have already written — repair it." >&2
fi
rm -f "$VALIDATE_ERR"
exit 1
fi
rm -f "$VALIDATE_ERR"
fi
# Preflight: recompute the browser marker from task.toml.
#
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
# by leaving it alone.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
if [ -d "$TASK_DIR/environment" ]; then
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
fi
# Preflight: restage the DNS jail script, for the same reason as the marker above.
#
# The Dockerfile COPYs environment/dns-jail/, and a missing COPY source fails the BUILD --
# which would kill every trial on the task, the one outcome the jail must never cause. A
# context staged before this existed passes the workspace guard above and would then die at
# build, so recreate the directory here and refill it when the source is around. Derived,
# so there is nothing to preserve by leaving it alone.
if [ -d "$TASK_DIR/environment" ]; then
mkdir -p "$TASK_DIR/environment/dns-jail"
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
cp "$DNSJAIL_SRC" "$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
break
fi
done
fi
# Preflight: report — never block — on edits to toolkit-managed files.
#
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt-consolidated.md ship from
# task-shared/ and decide how the trial runs and how the grade is produced, so an edit
# makes a task's runs hard to compare with the rest. Surface that here, before a trial
# burns agent time. It is advisory on purpose: an author who edited one did it to get
# unstuck, and refusing to run their trial punishes a misunderstanding. `|| true` also
# means a checker that can't run (a fresh unzip with no node_modules) never reads as an
# edit. The checker only exists in the worker toolkit; here the file is absent.
CHECK_INFRA="$REPO_ROOT/scripts/check-task-infra.ts"
if [ -f "$CHECK_INFRA" ]; then
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi
# Preflight: warn — loudly, but never block — when the live workspace has
# changes that a rebuild from the pinned commit + workspace.patch would lose.
# Trials run against the live workspace, but the finalized task keeps only the
# rebuild inputs (the workspace/ dir is gitignored) and every downstream
# consumer rebuilds from them, so anything uncaptured silently vanishes after
# packaging.
# The helper sits next to this script in a packed toolkit and under the
# toolkit's static scripts in the internal repo layout.
for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \
"$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do
if [ -f "$WS_SYNC" ]; then
bash "$WS_SYNC" "$TASK_DIR" || true
break
fi
done
# Agent + model selection, from scripts/harness-registry.toml via resolve_harness.
# The agent classes come from scripts/{snapshot,codex,gemini}_agent.py or
# harness_agents.py (hence the PYTHONPATH). The Claude variants reuse the claude
# binary baked into the task image instead of re-downloading it at agent-setup —
# stock claude-code's runtime download (~240 MB) races the 360s agent-setup timeout
# and loses on slow-egress hosts (AgentSetupTimeoutError). Tasks that ship a
# non-empty environment/session.jsonl additionally resume the staged session;
# single-turn tasks get the non-resuming class.
#
# resolve_harness exits non-zero (and prints why) when the selection could not
# produce a usable grade — an unknown/disabled harness, one that writes no ATIF
# trajectory, a missing credential, or a task needing resume on a harness that
# can't. Failing here costs a second; failing later costs the whole trial, and the
# resume case wouldn't fail at all, it would silently grade the wrong thing.
# _raccoon_python comes from lib/harness-credentials.sh, sourced above. Define a fallback
# only if that file was missing, so the error below is about the interpreter rather than an
# unbound function.
command -v _raccoon_python >/dev/null 2>&1 || _raccoon_python() { return 1; }
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
RACCOON_PY=$(_raccoon_python) || {
echo "harbor-run: ERROR — no python3.11+ with tomllib on PATH, so the harness registry" >&2
echo "harbor-run: cannot be read and the agent class cannot be resolved. Set" >&2
echo "harbor-run: RACCOON_PYTHON to an interpreter that has tomllib (3.11+)." >&2
exit 1
}
RESOLVED="$("$RACCOON_PY" "$SCRIPT_DIR/resolve_harness.py" \
--task-dir "$TASK_DIR" \
${HARNESS_ARGS[@]+"${HARNESS_ARGS[@]}"})" || exit 1
eval "$RESOLVED"
# --agent-import-path, not --agent: harbor 0.20 deprecates it but still accepts it, and it
# is the flag whose recorded shape (`agents[0].import_path`) every run on disk and every
# reader expects. --agent leaves import_path null and puts the class in `name`, which
# silently empties the agent field in published benchmark rows. Revisit if the pin moves.
AGENT_FLAGS="--agent-import-path $AGENT_IMPORT_PATH"
# Empty EFFORT_KWARG means "run the harness's native default config" — pass no
# effort kwarg at all rather than an empty one, which harbor would reject.
EFFORT_FLAGS=""
[ -n "$EFFORT_KWARG" ] && EFFORT_FLAGS="--ak $EFFORT_KWARG=$EFFORT_VALUE"
# FAST_KWARG is non-empty only when --fast was passed AND the harness declares one
# (resolve_harness refuses the flag otherwise).
FAST_FLAGS=""
[ -n "${FAST_KWARG:-}" ] && FAST_FLAGS="--ak $FAST_KWARG=true"
# The kwarg above reaches the trial agent only; the verifier launches its own grader
# claude, which the task's tests/test.sh puts in fast mode from GRADER_FAST_MODE.
GRADER_FAST_FLAGS=""
[ -n "$FAST_REQUESTED" ] && GRADER_FAST_FLAGS="--verifier-env GRADER_FAST_MODE=true"
# Environment backend. An explicit HARBOR_ENV always wins (either direction).
# Otherwise the default is context-dependent:
# - daytona for the internal repo: runs the trial in a cloud sandbox over
# HTTP, so it needs no local docker daemon and works *inside* the primary
# devcontainer. (HARBOR_ENV=docker uses the host docker daemon instead —
# free and offline, but host-only; the devcontainer has no docker.sock.)
# - docker for the worker toolkit: it's provisioned only for the local docker
# backend (docker CLI + bind-mounted docker.sock, no DAYTONA_API_KEY), so a
# daytona default would just error out. We detect a toolkit checkout by
# toolkit.json at the repo root — a file the packaging step writes that the
# internal repo never has. It lives in the bind-mounted workspace, not a
# baked image layer, so this holds even when the container's HARBOR_ENV pin
# is missing (e.g. a stale, pre-pin image).
if [ -n "${HARBOR_ENV:-}" ]; then
ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
ENV_TYPE="docker"
else
ENV_TYPE="daytona"
fi
# --no-delete keeps the environment around after the trial for inspection.
# That's free for a local docker container, but a Daytona or Modal sandbox is
# *billed* while it exists — keeping it would leak a paid sandbox on every run.
# Harbor downloads the trial logs into harbor-jobs before teardown either way, so
# for the cloud backends we let it delete the sandbox; for docker we keep the container.
DELETE_FLAGS="--no-delete"
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
# Orphan resilience (daytona): harbor tears sandboxes down per-trial + via an
# atexit that only closes the client — neither runs on SIGTERM/SIGKILL/crash, so
# a killed run leaks STARTED sandboxes that hog the shared pool until (if ever)
# an account default reaps them. Tell Daytona to auto-stop an IDLE sandbox after
# 20 min (auto-delete on stop), so orphans self-clean however the process dies.
# Safe for live trials: a running agent/grader keeps the sandbox active.
AUTOSTOP_FLAGS=""
[ "$ENV_TYPE" = "daytona" ] && AUTOSTOP_FLAGS="--ek auto_stop_interval_mins=20"
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
RESOURCE_FLAGS=""
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
exit 1
fi
# Task-image Claude Code floor. A task image installs Claude Code when it is first built and
# Docker reuses that layer on every later build, --force-build included, so an image built
# before the grader model's minimum CLI shipped fails every grade with "does not support this
# model" and the trial ends in RewardFileNotFoundError. Before a local docker trial, check the
# hb__ task images on this daemon and remove any that are too old, together with the build
# cache, so harbor's build below installs a current CLI. Advisory: no docker, no daemon, no
# images, or RACCOON_SKIP_IMAGE_PREFLIGHT=1 means nothing happens. The floor follows
# GRADER_CLI_MIN in the shared test.sh; RACCOON_CLAUDE_CODE_MIN overrides it.
CLAUDE_CODE_MIN="${RACCOON_CLAUDE_CODE_MIN:-}"
if [ -z "$CLAUDE_CODE_MIN" ]; then
for _ts in "$REPO_ROOT/task-shared/test.sh" "$REPO_ROOT/harbor-tasks/raccoon-shared/test.sh"; do
[ -f "$_ts" ] || continue
CLAUDE_CODE_MIN="$(sed -n 's/^GRADER_CLI_MIN=.*:-\([0-9][0-9.]*\)}.*/\1/p' "$_ts" | head -1)"
[ -n "$CLAUDE_CODE_MIN" ] && break
done
fi
CLAUDE_CODE_MIN="${CLAUDE_CODE_MIN:-2.1.251}"
if [ "$ENV_TYPE" = "docker" ] && [ "${RACCOON_SKIP_IMAGE_PREFLIGHT:-0}" != "1" ] \
&& command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
STALE_IMAGES=""
for img in $(docker images --format '{{.Repository}}:{{.Tag}}' 2>/dev/null | grep '^hb__' || true); do
v="$(docker run --rm --entrypoint claude "$img" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
[ -n "$v" ] || continue
if [ "$(printf '%s\n%s\n' "$CLAUDE_CODE_MIN" "$v" | sort -V | head -1)" != "$CLAUDE_CODE_MIN" ]; then
STALE_IMAGES="$STALE_IMAGES $img"
echo "Task image $img carries Claude Code $v; the grader needs $CLAUDE_CODE_MIN or newer. Removing it so the next build installs a current one." >&2
fi
done
if [ -n "$STALE_IMAGES" ]; then
for img in $STALE_IMAGES; do
# Trial containers harbor kept for inspection pin the image; drop them first.
for c in $(docker ps -aq --filter "ancestor=$img" 2>/dev/null); do docker rm -f "$c" >/dev/null 2>&1 || true; done
docker image rm -f "$img" >/dev/null 2>&1 || true
done
# The stale install layer also lives in the build cache, where a rebuild would find it.
docker builder prune -af >/dev/null 2>&1 || true
echo "The task image rebuilds once with a current Claude Code; later runs reuse it." >&2
fi
fi
# Launch-time input-checksum capture. Snapshot the task inputs BEFORE harbor
# starts, so the recorded hashes are what the agent actually ran against —
# an input edited between this run and copy-reference-run no longer records
# post-edit state and masks staleness. The capture is stamped into the trial
# dirs after the run (below); copy-reference-run prefers it over its weaker
# capture-at-copy fallback. Advisory end to end: a failure here never blocks
# the run.
#
# The post-run stamping watches the default `harbor-jobs` output dir. If the
# caller overrides the output dir via extra args, we can't know where the
# trials will land — skip stamping and say so, rather than silently stamping
# nothing (runs then fall back to copy-time capture in copy-reference-run).
STAMP_FILE=""
for arg in "$@"; do
case "$arg" in
-o|--output*)
echo "Note: custom harbor output dir passed ($arg) — skipping launch-time input-checksum stamping; reference runs will fall back to copy-time capture." >&2
STAMP_FILE="skip"
break
;;
esac
done
if [ "$STAMP_FILE" != "skip" ]; then
STAMP_FILE="$(mktemp)"
if ! npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" capture "$TASK_DIR" --out "$STAMP_FILE"; then
echo "Warning: could not capture launch-time input checksums (staleness will be judged from copy-time capture instead)" >&2
rm -f "$STAMP_FILE"
STAMP_FILE=""
fi
else
STAMP_FILE=""
fi
JOBS_BEFORE="$(ls -1 harbor-jobs 2>/dev/null || true)"
# DNS jail: the trial resolves the LLM proxy and nothing else. Opt-in with
# RACCOON_DNS_JAIL=1, which .env can set; the agent applies it inside the container from
# the proxy URL it already holds, so nothing is passed on the harbor command line.
#
# Claude and codex only (see scripts/dnsjail.py): both are baked into the task images,
# while every other harness downloads its CLI at agent-setup. Say so out loud — a silently
# unjailed trial is worse than no jail, because the caller believes otherwise.
if [ "${RACCOON_DNS_JAIL:-0}" = "1" ]; then
case "$AGENT_IMPORT_PATH" in
snapshot_agent:* | codex_agent:*)
# The script is baked into the image, so a task frozen before this feature has
# nothing to invoke. Most hand-authored per-task Dockerfiles are in that group.
if [ -f "$TASK_DIR/environment/Dockerfile" ] &&
! grep -q 'raccoon-dns-jail' "$TASK_DIR/environment/Dockerfile"; then
echo "Note: DNS jail skipped — this task's image predates it and ships no" \
"resolver. This trial has full network access." >&2
fi
;;
*) echo "Note: DNS jail skipped — it is applied by the claude and codex agents" \
"only. This trial has full network access." >&2 ;;
esac
fi
# The model comes from the registry (already resolved above) and is always a
# CONCRETE id, never a shorthand alias. On the manual (non-snapshot) path the agent
# passes the model via ANTHROPIC_MODEL, where a shorthand is NOT alias-resolved, so
# the configured base-URL endpoint rejects it (400 "Invalid model: <shorthand>") and
# trials die on turn 1. Bump `default_model` in scripts/harness-registry.toml when a
# newer model ships.
#
# harbor runs as a child (this script used to `exec` it, but the post-run
# stamping needs to run after harbor exits), so forward TERM/INT: a `kill`
# aimed at this wrapper's PID must take harbor down with it, not orphan a
# running job (this repo has been bitten by zombie harbor coordinators
# before).
HARBOR_EXIT=0
HARBOR_SIGNALLED=""
harbor run \
-p "$TASK_DIR" \
$AGENT_FLAGS \
-m "$MODEL" \
-e "$ENV_TYPE" \
$DELETE_FLAGS \
$AUTOSTOP_FLAGS \
$RESOURCE_FLAGS \
--yes \
-o harbor-jobs \
$EFFORT_FLAGS \
$FAST_FLAGS \
$GRADER_FAST_FLAGS \
"$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
wait "$HARBOR_PID" || HARBOR_EXIT=$?
if [ -n "$HARBOR_SIGNALLED" ]; then
# The first wait was interrupted by the trap; wait again so harbor's real
# exit status (not the shell's signal status) is what we propagate.
wait "$HARBOR_PID" || HARBOR_EXIT=$?
fi
trap - TERM INT
# Stamp the launch-time capture into this task's trial dirs in the job dir(s)
# this run created (the harbor-jobs entries that didn't exist before the
# run). Harbor names job dirs with a timestamp, so new entries are this run's
# output — plus, when several harbor-runs share a cwd, possibly a concurrent
# run's; `apply` is slug-scoped so another task's trials are never stamped
# with this task's inputs.
if [ -n "$STAMP_FILE" ]; then
NEW_JOBS="$(comm -13 <(printf '%s\n' "$JOBS_BEFORE" | sort) <(ls -1 harbor-jobs 2>/dev/null | sort) | sed 's|^|harbor-jobs/|')"
if [ -n "$NEW_JOBS" ]; then
# shellcheck disable=SC2086 # job-dir names are timestamps, never spaced
npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" apply "$STAMP_FILE" $NEW_JOBS || \
echo "Warning: could not stamp trial dirs with launch-time input checksums" >&2
fi
rm -f "$STAMP_FILE"
fi
# Repeat the toolkit-managed-file notice AFTER the trial. The preflight copy is
# minutes of harbor output up the scrollback by now, which for a notice nothing
# enforces means nobody reads it. This one lands where the author is looking.
if [ -f "$CHECK_INFRA" ]; then
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi
exit "$HARBOR_EXIT"

View File

@@ -0,0 +1,271 @@
version = 1
[[harness]]
id = "claude-code"
label = "Claude Code"
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
# given a browser can look at the screenshot it just took. Distinct classes with distinct
# names, because a different toolset is a different agent.
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
import_path_aliases = [
"snapshot_agent:FullToolsetSnapshotClaudeCode",
"snapshot_agent:FullToolsetPreinstalledClaudeCode",
"harbor.agents.installed.claude_code:ClaudeCode",
]
legacy_bare_model_rows = true
default_model = "claude-opus-5[1m]"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
fast_kwarg = "fast_mode"
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "claude"
install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash && break; echo \"claude install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
# No agent_config: claude reduces its toolset with `--tools`, not `-c key=value`, so the
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
# unlike model and effort, which are interpolated from this row.
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
[[harness]]
id = "codex"
label = "OpenAI Codex CLI"
agent_import_path = "codex_agent:NativeSnapshotCodex"
agent_import_path_single_turn = "codex_agent:SystemNodeCodex"
import_path_aliases = [
"codex_agent:InlineSnapshotCodex",
"harbor.agents.installed.codex:Codex",
]
legacy_bare_model_rows = true
default_model = "gpt-5.6-sol"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
key_env = "OPENAI_API_KEY"
base_url_env = "OPENAI_BASE_URL"
proxy_path = "openai/v1"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "codex"
install = "for i in 1 2 3; do curl -fsSL https://chatgpt.com/codex/install.sh | CODEX_NON_INTERACTIVE=1 sh && break; echo \"codex install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
skills_dir = "$HOME/.agents/skills"
config_path = "${CODEX_HOME:-$HOME/.codex}/config.toml"
auth_path = "${CODEX_HOME:-$HOME/.codex}/auth.json"
auth_key_env = "OPENAI_API_KEY"
agent_config = """
web_search = "disabled"
[agents]
enabled = false
[tools]
update_plan = { enabled = false }
experimental_request_user_input = { enabled = false }
[features]
goals = false
multi_agent = false
multi_agent_v2 = false
memories = false
external_agent_memory_import = false
"""
container_config = """
openai_base_url = "${OPENAI_BASE_URL}"
"""
explore_config = """
[hooks]
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
"""
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
[[harness]]
id = "gemini-cli"
label = "Gemini CLI"
agent_import_path = "gemini_agent:NativeSnapshotGeminiCli"
agent_import_path_single_turn = "gemini_agent:SystemNodeGeminiCli"
import_path_aliases = ["harbor.agents.installed.gemini_cli:GeminiCli"]
legacy_bare_model_rows = true
default_model = "gemini-3.5-flash"
model_id_shape = "provider/model"
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GEMINI_API_BASE"
proxy_path = "gemini"
writes_atif = true
capture = false
seed_native = true
seed_atif = false
[[harness]]
id = "antigravity-cli"
label = "Antigravity CLI"
agent_import_path = "harness_agents:BenchAntigravity"
import_path_aliases = ["harbor.agents.installed.antigravity_cli:AntigravityCli"]
legacy_bare_model_rows = false
# The prefix is load-bearing: harbor's adapter raises without a "/" in the id.
# agy carries its own model catalogue and DROPS entries between point releases
# (1.1.25 removed gemini-3.5-flash, breaking every run). If trials start failing
# with "not recognized as a known model", run `agy --model bogus --prompt=x` to
# print the current catalogue and update this.
default_model = "google/gemini-3.8-flash"
model_id_shape = "provider/model"
# Not optional: agy refuses a Gemini 3 model with no --effort ("requires --effort
# (available: low, medium, high)"). low/high are safe on pro and flash alike.
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GOOGLE_GEMINI_BASE_URL"
proxy_path = "gemini"
writes_atif = true
capture = false
# agy cannot be handed externally-produced history, so multi-turn tasks must
# hard-fail rather than silently run cold. See work-logs/antigravity-harness.md.
seed_native = false
seed_atif = false
[[harness]]
id = "opencode"
label = "OpenCode"
agent_import_path = "harness_agents:BenchOpenCode"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "goose"
label = "Goose"
agent_import_path = "harness_agents:BenchGoose"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "mini-swe-agent"
label = "mini-swe-agent"
agent_import_path = "harness_agents:BenchMiniSweAgent"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "cline-cli"
label = "Cline CLI"
agent_import_path = "harness_agents:BenchCline"
legacy_bare_model_rows = false
model_id_shape = "provider:model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "crush"
label = "Crush"
agent_import_path = "harness_agents:Crush"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "amp"
label = "Amp"
agent_import_path = "harness_agents:Amp"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "AMP_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "cursor-cli"
label = "Cursor CLI"
agent_import_path = "harness_agents:BenchCursorCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "CURSOR_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "copilot-cli"
label = "GitHub Copilot CLI"
agent_import_path = "harness_agents:BenchCopilotCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "GITHUB_TOKEN"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "aider"
label = "Aider"
agent_import_path = "harness_agents:BenchAider"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
writes_atif = false
capture = false
seed_native = false
seed_atif = false
enabled = false

View File

@@ -0,0 +1,29 @@
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
// real shape instead of `any`.
export interface Turn {
/** Line index in the native session file. */
index: number;
role: 'user' | 'assistant';
text: string;
/** A slash-command turn, not real conversation. */
isCommand: boolean;
/** This record concluded its turn — the truncation boundary. */
endsTurn: boolean;
}
export interface Session {
harness: string;
rawPath: string;
/** The harness own id for this conversation. */
sessionId: string | null;
lines: string[];
turns: Turn[];
}
export function supportedHarnesses(): string[];
export function readSession(harness: string, recordedPath?: string): Session | null;
export function truncationIndex(turns: Turn[]): number;
export function turnsFromLines(harness: string, lines: string[]): Turn[];
export function linearSnapshotLines(session: Session, startLine?: number): string[];
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];

View File

@@ -0,0 +1,325 @@
// Locate and read a harness's native conversation, so capture-snapshot can work
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
// annotation, metadata) is harness-agnostic.
//
// The returned session stays in the harness's OWN native format: the seeding design
// hands a native blob back to the same harness, and codex_agent reads the same staged
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* @typedef {object} Turn
* @property {number} index line index in the native session file
* @property {'user'|'assistant'} role
* @property {string} text
* @property {boolean} isCommand a slash-command turn, not real conversation
* @property {boolean} endsTurn this record concluded its turn
*/
/**
* @typedef {object} Session
* @property {string} harness
* @property {string} rawPath
* @property {string|null} sessionId the harness's own id for this conversation
* @property {string[]} lines
* @property {Turn[]} turns
*/
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
// answer is stable — two sessions written in the same millisecond are common.
function newestUnder(root, matches) {
if (!fs.existsSync(root)) return null;
const found = [];
const walk = (dir) => {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
}
};
walk(root);
if (found.length === 0) return null;
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
return found[0].full;
}
// User-role records codex writes that the human did not type: its own environment
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
// Matched only at the START of the text, so a turn that merely quotes one is still real
// conversation.
function isCodexCommandText(text) {
const trimmed = (text || '').trimStart();
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
return /^\$[\w:.-]+\s*$/.test(trimmed);
}
const HARNESSES = {
'claude-code': {
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
findSession() {
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
n.endsWith('.jsonl')
);
},
/** Claude names the transcript for its session id. */
sessionId(rawPath) {
return path.basename(rawPath, '.jsonl');
},
/**
* One turn per conversational record. `endsTurn` marks an assistant record that
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
* file-history, permission-mode) carry no role and are skipped.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let entry;
try {
entry = JSON.parse(line);
} catch {
continue;
}
const role =
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
if (!role) continue;
const content = entry.message?.content;
const text =
typeof content === 'string'
? content
: Array.isArray(content)
? content
.filter((b) => b && b.type === 'text')
.map((b) => b.text ?? '')
.join('')
: '';
turns.push({
index,
role,
text,
isCommand:
role === 'user' &&
typeof content === 'string' &&
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
});
}
return turns;
},
},
codex: {
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
findSession() {
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
return newestUnder(
path.join(home, 'sessions'),
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
);
},
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
sessionId(rawPath, lines) {
for (const line of lines) {
try {
const rec = JSON.parse(line);
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
} catch {
continue;
}
}
return null;
},
/**
* codex rollouts carry `response_item` records whose payload is a message with a
* role. An assistant message with no following tool activity ends the turn; codex
* records no stop_reason, so a turn ends where the next user message begins —
* resolved after the fact below.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let record;
try {
record = JSON.parse(line);
} catch {
continue;
}
if (record.type !== 'response_item') continue;
const payload = record.payload ?? {};
if (payload.type !== 'message') continue;
const role =
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
if (!role) continue;
const text = Array.isArray(payload.content)
? payload.content.map((b) => b?.text ?? '').join('')
: typeof payload.content === 'string'
? payload.content
: '';
turns.push({
index,
role,
text,
isCommand: role === 'user' && isCodexCommandText(text),
endsTurn: false,
});
}
// An assistant turn ends where the next user turn starts, or at the end.
for (let i = 0; i < turns.length; i += 1) {
if (turns[i].role !== 'assistant') continue;
const next = turns[i + 1];
turns[i].endsTurn = !next || next.role === 'user';
}
return turns;
},
},
};
/** @returns {string[]} */
export function supportedHarnesses() {
return Object.keys(HARNESSES);
}
/**
* Read the current session for `harness`. Returns null when nothing is found, so the
* caller can report which harness had no conversation to capture.
*/
/**
* @param {string} harness
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
* over the newest-file scan, which can pick a different session in a busy container.
* @returns {Session | null}
*/
export function readSession(harness, recordedPath) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
if (!rawPath) return null;
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
return {
harness,
rawPath,
lines,
turns: reader.readTurns(lines),
sessionId: reader.sessionId(rawPath, lines),
};
}
/**
* Index of the last record to keep: the last turn-ending assistant record before the
* final real user turn. Drops the prompt that elicited the failure and the failure
* response, so the test agent inherits context but not the answer.
*
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
* treat as "seed nothing and run cold".
*/
/**
* @param {Turn[]} turns
* @returns {number}
*/
export function truncationIndex(turns) {
let lastUser = -1;
for (const turn of turns) {
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
}
if (lastUser < 0) return -1;
let cut = -1;
for (const turn of turns) {
if (turn.index >= lastUser) break;
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
}
return cut;
}
/**
* Parse already-read lines with a harness's reader, for callers that have the text
* rather than a path.
*
* @param {string} harness
* @param {string[]} lines
* @returns {Turn[]}
*/
export function turnsFromLines(harness, lines) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
return reader.readTurns(lines);
}
/**
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
* everything up to the snapshot invocation, matching what Claude Code stages when it
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
* in snapshot-to-task — capture keeps the full conversation.
*
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
* the whole session is kept, which would include the snapshot's own Q&A.
*
* @param {Session} session
* @param {number} [startLine]
* @returns {string[]}
*/
export function linearSnapshotLines(session, startLine) {
if (typeof startLine === 'number' && startLine >= 0) {
return session.lines.slice(0, startLine);
}
return session.lines;
}
/**
* Drop records that describe the AUTHORING container rather than the conversation.
*
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
* so without this the test agent inherits a list of skills it does not have — one described
* as "capture the current conversation and repo state as a snapshot" — and a working
* directory that does not exist in the trial. codex re-injects both for the trial, and base
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
* already re-records with the trial's own cwd; this brings codex to the same place.
*
* @param {string} harness
* @param {string[]} lines
* @returns {string[]}
*/
export function stripAuthoringScaffolding(harness, lines) {
if (harness === 'claude-code') return lines;
return lines.filter((raw) => {
let rec;
try {
rec = JSON.parse(raw);
} catch {
return true;
}
const payload = rec?.payload;
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
const text = (payload.content ?? [])
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
.join('')
.trim();
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
// worker who quotes one of these strings mid-conversation keeps their turn.
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
if (payload.role === 'user') return !text.startsWith('<environment_context>');
return true;
});
}

View File

@@ -0,0 +1,19 @@
import { existsSync } from 'node:fs';
(function checkDevcontainer() {
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
if (inContainer) return;
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
if (process.env.CI === 'true' || process.env.CI === '1') return;
const yellow = '\x1b[33m';
const reset = '\x1b[0m';
process.stderr.write(
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
);
})();

View File

@@ -0,0 +1,36 @@
"""How codex is handed its API key, kept out of codex_agent so it is testable without
harbor (whose venv has no pytest, so anything importing it SKIPs in CI).
codex reads its key from `$CODEX_HOME/auth.json` and its proxy URL from config.toml —
`OPENAI_API_KEY` / `OPENAI_BASE_URL` in the environment are both ignored, verified against
0.146.0 and 0.152.0 (an env-var-only run sends no `authorization` header at all).
"""
from __future__ import annotations
import json
import shlex
# Characters harbor's own auth.json writer cannot survive: it interpolates the key into a
# shell heredoc, so `"` closes the JSON string and `\` starts an escape.
_UNESCAPABLE = '"\\\n\r'
AUTH_JSON_ENV_VAR = "RACCOON_CODEX_AUTH_JSON"
def auth_json_setup(key: str, remote_auth_path: str) -> tuple[dict[str, str], str]:
"""The one extra env var — returned separately so it reaches ONLY the setup exec — plus
shell writing a parseable auth.json. Subshell: the umask must not outlive this write."""
env = {AUTH_JSON_ENV_VAR: json.dumps({"OPENAI_API_KEY": key})}
command = (
f"(umask 077; printf '%s\\n' \"${AUTH_JSON_ENV_VAR}\" "
f">{shlex.quote(remote_auth_path)})\n"
)
return env, command
def unescapable_chars(key: str) -> list[str]:
"""Which characters in `key` harbor's stock heredoc writer would corrupt — empty for
every ordinary key, so the caller can refuse instead of 401ing three layers down."""
return sorted({c for c in _UNESCAPABLE if c in key})

View File

@@ -0,0 +1,43 @@
/**
* Recursive copy for scripts that must not call `cpSync`: it fails EACCES
* against a macOS docker bind mount, where the toolkit's job dirs live.
*/
import {
chmodSync,
copyFileSync,
lstatSync,
mkdirSync,
readdirSync,
readlinkSync,
rmSync,
statSync,
symlinkSync,
} from 'fs';
import { join } from 'path';
/** Copy one entry — symlink, directory or file — preserving its mode. */
export function copyPath(src: string, dest: string) {
const st = lstatSync(src);
if (st.isSymbolicLink()) {
rmSync(dest, { force: true });
symlinkSync(readlinkSync(src), dest);
return;
}
if (st.isDirectory()) {
copyTree(src, dest);
return;
}
// Unlink first: copyFileSync onto an existing file keeps that file's mode.
rmSync(dest, { force: true });
copyFileSync(src, dest);
chmodSync(dest, statSync(src).mode & 0o777);
}
/** Copy `src`'s contents into `dest`, creating `dest` if it doesn't exist. */
export function copyTree(src: string, dest: string) {
mkdirSync(dest, { recursive: true });
for (const entry of readdirSync(src, { withFileTypes: true })) {
copyPath(join(src, entry.name), join(dest, entry.name));
}
}

View File

@@ -0,0 +1,137 @@
#!/bin/sh
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
# every other name unresolvable. Runs as root, inside the container.
#
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
#
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
# applied before it is verified, and any doubt leaves the container's DNS untouched.
set -u
STATE=/tmp/.dnsjail
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
# later run could mistake for its own filter.
drop_ours() {
if [ -s "$STATE/dnsmasq.pid" ]; then
pid=$(cat "$STATE/dnsmasq.pid")
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
# some service's child. Confirm it is dnsmasq before signalling it.
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
dnsmasq) kill "$pid" 2>/dev/null || true ;;
esac
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
fi
}
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
# end the caller's shell.
dnsjail_apply() {
required="${DNSJAIL_ALLOW:-}"
extra="${DNSJAIL_ALLOW_EXTRA:-}"
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
# A blank required list means no model endpoint was found: jailing would strand the agent.
set -- $required
[ $# -gt 0 ] || return 0
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
# silently UNjail a working container.
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
return 0
fi
# The state dir has to work first: it holds what unjail restores, and a failed write here
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
# running as the container user in Explore, can drop its own lift markers.
mkdir -p "$STATE" 2>/dev/null || return 0
chmod 1777 "$STATE" 2>/dev/null || true
: > "$STATE/.probe" 2>/dev/null || return 0
rm -f "$STATE/.probe" 2>/dev/null || true
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
# every name.
src=/etc/resolv.conf
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
[ "$up" = "127.0.0.1" ] && up=""
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
srv=""
for h in $allow; do srv="$srv --server=/$h/$up"; done
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi
# Ask the resolver directly: the model endpoint must answer and the control must not --
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
# through the catch-all, and one of those must not silently disable the whole jail.
live=1
for h in $required; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
done
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
# resolve through the catch-all, and must not take the whole jail down with it.
if [ -n "$live" ]; then
for h in $extra; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
done
fi
if [ -z "$live" ]; then
# Say why. A silent decline is indistinguishable from a jail that worked, and the
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
# AF_NETLINK, so dnsmasq cannot start there at all).
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
drop_ours
# Failing open has to mean actually open, including when an earlier run left this
# container jailed.
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
fi
return 0
fi
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
# would leave unjail a permanent no-op.
if ! jailed_now; then
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
fi
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
rm -rf "$STATE/lifts" 2>/dev/null || true
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
# which means the replacement has to be complete BEFORE the write starts. Keep every
# non-nameserver directive docker set (options, search).
{ printf 'nameserver 127.0.0.1\n'
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
} > "$STATE/resolv.jailed" 2>/dev/null
[ -s "$STATE/resolv.jailed" ] || return 0
cat "$STATE/resolv.jailed" > /etc/resolv.conf
}
dnsjail_apply || true

View File

@@ -0,0 +1,263 @@
#!/bin/bash
# Read the harness registry and derive per-harness credentials from it.
#
# Source it — the whole point is exporting into the caller's environment, which a subshell
# would lose:
#
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
# harness_setup_credentials
#
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
# re-derives and rewrites the auth files before an interactive launch; and
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
# on top.
#
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
# post-creates run with -e). An unguarded failure below therefore aborts container
# creation, which is why every failure site is individually guarded rather than relying on
# this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
# the first one that can actually import it rather than assuming.
_raccoon_python() {
local p
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
[ -n "$p" ] || continue
command -v "$p" >/dev/null 2>&1 || continue
if "$p" -c "import tomllib" >/dev/null 2>&1; then
printf '%s' "$p"
return 0
fi
done
return 1
}
_harness_query() {
local py
py=$(_raccoon_python) || return 1
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
}
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
# whitespace anywhere, so deleting rather than trimming needs no cases.
_harness_trim() {
local out
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
printf '%s' "${out:-$1}"
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
[ -n "$base_url" ] || return 1
base_url="${base_url%"${base_url##*[!/]}"}"
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
# URL that is a bare host with no path — a provider's own API root rather than the
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
case "${base_url#*://}" in
*/*) printf '%s' "${base_url%/*}" ;;
*) return 2 ;;
esac
}
harness_setup_credentials() {
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
# note at the top), and a bare failing assignment would exit the caller's post-create
# outright — silently, since the failure paths below are what do the explaining.
local root rc=0
root="$(_harness_proxy_root)" || rc=$?
if [ "$rc" -ne 0 ]; then
if [ "$rc" -eq 2 ]; then
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
echo "harness-setup: authenticated. Use the base URL you were given." >&2
else
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
fi
return 0
fi
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
export ANTHROPIC_BASE_URL
local key
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
if [ -z "$key" ]; then
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
return 0
fi
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
export ANTHROPIC_API_KEY="$key"
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
if [ -z "${!key_env:-}" ]; then
export "$key_env=$key"
fi
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
export "$base_url_env=$root/$proxy_path"
fi
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
harness_write_auth() {
local id auth_path key_env target key py
py=$(_raccoon_python) || {
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
return 0
}
while IFS=$'\t' read -r id auth_path key_env; do
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
# too — this is the value that reaches the file the harness authenticates with.
key="$(_harness_trim "${!key_env:-}")"
if [ -z "$key" ]; then
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
continue
fi
target=$(eval "printf '%s' \"$auth_path\"") || {
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")" || {
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
continue
}
# json.dumps, not printf: a key containing a quote or backslash would otherwise
# produce a file the CLI cannot parse, and the failure would surface as an auth
# error rather than a malformed file.
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
"$py" -c 'import json, os
target = os.environ["RACCOON_AUTH_TARGET"]
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
fh.write("\n")
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
continue
fi
echo "harness-setup: $id auth -> $target" >&2
done < <(_harness_query --auth-files 2>/dev/null || true)
}
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
# leaving every other line — the explore surface's [hooks] table included — untouched.
harness_refresh_config_keys() {
local id config_path blob target py
py=$(_raccoon_python) || return 0
# The surface only decides what a CREATE writes. An update takes the root keys off the
# front of the same blob, so a surface's tables survive byte-for-byte either way.
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
target=$(eval "printf '%s' \"$config_path\"") || continue
mkdir -p "$(dirname "$target")" || continue
if printf '%s' "$blob" | base64 -d |
RACCOON_CONFIG_TARGET="$target" "$py" -c '
import os, re, sys, tomllib
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
target = os.environ["RACCOON_CONFIG_TARGET"]
text = sys.stdin.read()
# Empty counts as unresolved: writing an empty base URL would break a container whose
# config is currently right, which is the one thing this must never do.
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
raise SystemExit(1)
text = os.path.expandvars(text)
wanted = []
for line in text.splitlines():
if line.lstrip().startswith("["):
break
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
if m:
wanted.append((m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
mode = None
if os.path.exists(target):
try:
with open(target, encoding="utf-8") as fh:
lines = fh.read().splitlines()
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
# Everything from the first table header on belongs to a table. A key appended after
# one is reparented into it, so both the search and the insert stay above the line.
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
changed = False
for key, line in wanted:
# The quoted spelling is the same key: replacing it beats adding a duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
if at is None:
if root_end < len(lines) and lines[root_end].strip():
lines.insert(root_end, "")
lines.insert(root_end, line)
root_end += 1
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
else:
# No file means container-create could not write one, so write what it would have:
# on the explore surface that is the capture hooks too, not just the root keys.
out = HEADER + "\n" + text
try:
doc = tomllib.loads(out)
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have reached the root.
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with open(tmp, "w", encoding="utf-8") as fh:
fh.write(out)
if mode is not None:
os.chmod(tmp, mode)
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: $id config keys refreshed -> $target" >&2
fi
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}

View File

@@ -0,0 +1,418 @@
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
package.json has no zod/smol-toml).
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
copies.
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
sandbox agent, a plain unit test, or the devcontainer python alike.
"""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
@dataclass(frozen=True)
class Harness:
"""One harness, as declared in harness-registry.toml."""
id: str
label: str
agent_import_path: str
model_id_shape: str
writes_atif: bool
capture: bool
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
# has no `Read` equivalent to switch toolsets for.
agent_import_path_browser: str | None = None
agent_import_path_single_turn_browser: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
effort_kwarg: str = ""
effort_default: str | None = None
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
fast_kwarg: str = ""
key_env: str | None = None
base_url_env: str | None = None
proxy_path: str | None = None
flaky_hangs: bool = False
enabled: bool = True
# Worker-container fields; see the registry header.
authoring: bool = False
cli: str | None = None
install: str | None = None
skills_dir: str | None = None
auth_path: str | None = None
auth_key_env: str | None = None
explore_launch: str | None = None
config_path: str | None = None
# Config the harness needs wherever it runs, trial sandbox included.
agent_config: str | None = None
# Config for both worker containers (explore and authoring).
container_config: str | None = None
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
# explore/plugins/. Writing them in authoring would register hooks against files that
# are not there, firing on every prompt.
explore_config: str | None = None
# Fields added for a later phase, kept verbatim so this loader doesn't have to
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there.
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
built-in — a different toolset is a different agent, so it is a different class with
its own name rather than a flag on the canonical one. Harnesses without a variant fall
through to their normal class."""
if browser:
variant = (
self.agent_import_path_browser
if multi_turn
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
)
if variant:
return variant
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
def row_label(self, model: str) -> str:
"""Row identity for one trial: bare model for legacy harnesses (so
published manifests keep their labels), else ``<harness>:<model>``."""
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
def agent_config_overrides(self) -> dict[str, str]:
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
interpolates these into a shell command, which would strip the quotes anyway;
emitting them would only make the result depend on how many shell layers the
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
binary as ``model=o3``).
These settings ride the command line as ``-c dotted.key=value`` everywhere the
harness runs, never a config file. A trial sandbox rules the file out: the
harness's own runner appends root keys to it, and TOML has no way back to the
root scope once a table has opened, so a table we appended would swallow them.
Overrides compose in any order and beat the file, so the same rendering serves
the explore launcher too — one declaration, one mechanism.
"""
if not self.agent_config:
return {}
try:
parsed = tomllib.loads(self.agent_config)
except tomllib.TOMLDecodeError as exc:
raise HarnessRegistryError(
f"{self.id}: agent_config is not valid TOML ({exc})"
) from exc
flat: dict[str, str] = {}
def walk(node: dict, prefix: str) -> None:
for key, value in node.items():
path = f"{prefix}{key}"
if isinstance(value, dict):
walk(value, f"{path}.")
elif isinstance(value, bool):
flat[path] = "true" if value else "false"
elif isinstance(value, (int, float)):
flat[path] = str(value)
elif isinstance(value, str):
if value != value.strip() or any(c in value for c in " \"'\\"):
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has a value needing "
"shell quoting, which the -c override form cannot carry"
)
flat[path] = value
else:
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has type "
f"{type(value).__name__}, which has no -c override form"
)
walk(parsed, "")
return flat
def container_config_text(self, *, surface: str) -> str | None:
"""Config file body for a worker container. `surface` is "explore" or
"authoring"; explore additionally gets `explore_config`. Root keys come from
`container_config` first, so appending a table section stays valid TOML."""
parts = [self.container_config]
if surface == "explore":
parts.append(self.explore_config)
kept = [part.strip("\n") for part in parts if part and part.strip()]
return "\n\n".join(kept) + "\n" if kept else None
def agent_config_flags(self) -> str:
"""``agent_config`` as a ``-c key=value`` command-line string."""
return " ".join(
f"-c {key}={value}"
for key, value in sorted(self.agent_config_overrides().items())
)
def explore_launch_command(self) -> str | None:
"""``explore_launch`` with the registry's own values substituted in.
The worker's Explore session and the trial must run the same agent, so the
model, effort and reductions are declared once here and rendered into both.
A literal in the launch string would be a second declaration, and the two
would drift the first time one of them was updated alone.
Only these three placeholders are substituted; ``$@`` and
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
"""
if not self.explore_launch:
return None
return (
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
.replace("$RACCOON_MODEL", self.default_model or "")
.replace("$RACCOON_EFFORT", self.effort_default or "")
)
def known_import_paths(self) -> tuple[str, ...]:
paths = [self.agent_import_path, *self.import_path_aliases]
if self.agent_import_path_single_turn:
paths.append(self.agent_import_path_single_turn)
return tuple(paths)
_KNOWN_FIELDS = frozenset(
{
"id",
"label",
"agent_import_path",
"agent_import_path_single_turn",
"agent_import_path_browser",
"agent_import_path_single_turn_browser",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
"model_id_shape",
"effort_kwarg",
"effort_default",
"fast_kwarg",
"key_env",
"base_url_env",
"proxy_path",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
"flaky_hangs",
"enabled",
"authoring",
"cli",
"install",
"skills_dir",
"auth_path",
"auth_key_env",
"explore_launch",
"config_path",
"agent_config",
"container_config",
"explore_config",
}
)
_REQUIRED_FIELDS = (
"id",
"label",
"agent_import_path",
"model_id_shape",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
)
class HarnessRegistryError(ValueError):
"""Malformed registry. Raised rather than tolerated: a broken registry is a
broken deployment, and silently defaulting would pick the wrong agent."""
def _references_agent_flags(launch: str) -> bool:
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
@dataclass(frozen=True)
class HarnessRegistry:
version: int
harnesses: tuple[Harness, ...]
def all(self) -> tuple[Harness, ...]:
return self.harnesses
def enabled(self) -> tuple[Harness, ...]:
return tuple(h for h in self.harnesses if h.enabled)
def authoring(self) -> tuple[Harness, ...]:
"""Harnesses a worker can author with — what the worker containers install.
Narrower than enabled(): a harness can be runnable in a trial without having
an authoring story (no CLI to converse with, or no capture)."""
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
def find(self, harness_id: str) -> Harness | None:
return next((h for h in self.harnesses if h.id == harness_id), None)
def require(self, harness_id: str) -> Harness:
harness = self.find(harness_id)
if harness is not None:
return harness
available = ", ".join(sorted(h.id for h in self.enabled()))
raise HarnessRegistryError(
f'Unknown harness "{harness_id}". Available: {available}'
)
def by_import_path(self, agent: str) -> Harness | None:
"""Resolve an agent identity — a ``name()`` or import path from
``result.json`` ``config.agent``, or a manifest row — to its harness."""
needle = (agent or "").strip()
if not needle:
return None
for harness in self.harnesses:
if needle == harness.id or needle in harness.known_import_paths():
return harness
return None
def _build(entry: dict, index: int) -> Harness:
for name in _REQUIRED_FIELDS:
if name not in entry:
raise HarnessRegistryError(
f"harness[{index}]: missing required field '{name}'"
)
shape = entry["model_id_shape"]
if shape not in MODEL_ID_SHAPES:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
f"{sorted(MODEL_ID_SHAPES)}"
)
# These three reach `eval` in setup-harnesses.sh, which is how they support the
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
# is ours, but "ours" is not an argument that survives a careless future edit.
for shell_field in ("config_path", "auth_path", "skills_dir"):
value = entry.get(shell_field)
if not isinstance(value, str):
continue
# A backtick or $( executes outright. A double quote closes the string these are
# interpolated into, and a semicolon then starts a new command inside it — same
# outcome, one step removed.
bad = [t for t in ("`", "$(", '"', ";") if t in value]
if bad:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): {shell_field} contains "
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
)
launch = entry.get("explore_launch")
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): declares agent_config but its "
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
"would then run with a different toolset than the trial it is authoring "
"for, which is the drift agent_config exists to prevent."
)
return Harness(
id=entry["id"],
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
agent_import_path_browser=entry.get("agent_import_path_browser"),
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),
model_id_shape=shape,
effort_kwarg=entry.get("effort_kwarg", ""),
effort_default=entry.get("effort_default"),
fast_kwarg=entry.get("fast_kwarg", ""),
key_env=entry.get("key_env"),
base_url_env=entry.get("base_url_env"),
proxy_path=entry.get("proxy_path"),
writes_atif=bool(entry["writes_atif"]),
capture=bool(entry["capture"]),
seed_native=bool(entry["seed_native"]),
seed_atif=bool(entry["seed_atif"]),
flaky_hangs=bool(entry.get("flaky_hangs", False)),
enabled=bool(entry.get("enabled", True)),
authoring=bool(entry.get("authoring", False)),
cli=entry.get("cli"),
install=entry.get("install"),
skills_dir=entry.get("skills_dir"),
auth_path=entry.get("auth_path"),
auth_key_env=entry.get("auth_key_env"),
explore_launch=entry.get("explore_launch"),
config_path=entry.get("config_path"),
agent_config=entry.get("agent_config"),
container_config=entry.get("container_config"),
explore_config=entry.get("explore_config"),
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
)
_cache: dict[Path, HarnessRegistry] = {}
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
file, a duplicate id, or an import path claimed by two harnesses (which would
make ``by_import_path`` depend on declaration order)."""
resolved = Path(path).resolve()
if resolved in _cache:
return _cache[resolved]
with open(resolved, "rb") as handle:
doc = tomllib.load(handle)
if "version" not in doc:
raise HarnessRegistryError("harness-registry: missing 'version'")
entries = doc.get("harness") or []
if not entries:
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
seen_ids: set[str] = set()
for harness in harnesses:
if harness.id in seen_ids:
raise HarnessRegistryError(
f"harness-registry: duplicate harness id: {harness.id}"
)
seen_ids.add(harness.id)
owners: dict[str, str] = {}
for harness in harnesses:
for import_path in harness.known_import_paths():
owner = owners.get(import_path)
if owner is not None and owner != harness.id:
raise HarnessRegistryError(
f'harness-registry: import path "{import_path}" claimed by both '
f'"{owner}" and "{harness.id}"'
)
owners[import_path] = harness.id
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
_cache[resolved] = registry
return registry

View File

@@ -0,0 +1,324 @@
/**
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
* that reference runs and detector reports depend on.
*
* A reference run is only meaningful for the task inputs it actually ran
* against: the prompt (instruction.md), the snapshot session
* (environment/session.jsonl), the workspace patch
* (environment/workspace.patch), and the gitref the workspace is built from
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
* revision of instruction.md + the task's holistic rubric
* (tests/holistic-rubric.md on current tasks; tests/grader-guidance-consolidated.md
* on tasks created before the rename) — and, for the rubric detectors, the
* task's atomic rubric (tests/atomic-rubric.yaml; tests/rubrics.yaml on tasks
* converted before the rename) and tests/grader-context.md. When any of those
* change after the artifact was produced, the artifact is stale — it describes
* an older revision of the task than the one being packaged.
*
* This module is the single source of truth for WHAT gets checksummed and how
* captures are compared. Capture sites (copy-reference-run.ts,
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
* the artifact; submit-task.ts re-captures at packaging time and diffs.
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
* mask an mtime, but can't change a sha256.
*
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
* occupies the same relative path there (mirroring the check-devcontainer
* pattern).
*
* Related but deliberately separate: `computeDeliveryHash` (repo-side
* delivery script — grep the internal repo for it; not shipped with the
* toolkit) hashes an overlapping input set for delivery idempotency. It is
* NOT built on this module because its hash format is load-bearing (a
* changed hash re-delivers every task); if you change WHAT counts as a task
* input here, check whether the delivery hash needs the same change.
*/
import { createHash } from 'node:crypto';
import { existsSync, readFileSync } from 'node:fs';
import { join } from 'node:path';
/** Bump when the record shape changes incompatibly. */
export const INPUT_CHECKSUMS_VERSION = 1;
/** Filename of the record inside a reference-run directory. */
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
/**
* Where in the artifact lifecycle a capture happened. The moment matters for
* how much a "fresh" verdict can be trusted:
*
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
* The strongest evidence: the record is what the agent ran
* against, whatever got edited afterwards.
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
* no run-time stamp. An input edited between harbor-run and the
* copy is recorded at its post-edit state, so a stale run can
* read fresh.
* - 'stamp' — right after a detector report is written
* (record-detector-inputs.ts in a worker checkout; the
* internal repo's detector save path stamps the same way).
* - 'mirror' — retired: written by the repo-side flow that re-materialized
* canonical detector reports to disk back when reports had a
* remote canonical store. Reports are local-only now, so no
* current code writes it; the member stays so old stamps keep
* their recorded method when read.
* - 'regrade' — a re-grade of an existing run. Present on records already on
* disk; no current code path writes it.
*
* Absent on records written before this field existed.
*/
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror' | 'regrade';
/** Runtime mirror of {@link TaskInputCaptureMethod}, for validating a record read from disk. */
export const TASK_INPUT_CAPTURE_METHODS = Object.freeze([
'run',
'copy',
'stamp',
'mirror',
'regrade',
] as const satisfies readonly TaskInputCaptureMethod[]);
/**
* The checksums of a task's inputs as they stood at capture time. Every hash
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
* that later becomes a hash — or vice versa — is a change like any other).
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
* than hashed so a mismatch message can show it.
*/
export interface TaskInputChecksums {
readonly version: number;
/** ISO-8601 timestamp of the capture. */
readonly capturedAt: string;
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
readonly capturedBy?: TaskInputCaptureMethod;
/**
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
* by launch-time captures so stamping can be scoped to the right task's
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
* stamp is detectable after the fact.
*/
readonly taskSlug?: string;
/**
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
* after the harbor-path scrub deliberately rewrote those docs
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
* task; only the doc bytes were normalized.
*/
readonly restampedAt?: string;
readonly inputs: {
readonly prompt: string | null;
readonly graderGuidance: string | null;
readonly sessionJsonl: string | null;
readonly workspacePatch: string | null;
readonly gitref: string | null;
/**
* tests/grader-guidance-consolidated.md — the holistic rubric under its
* pre-rename filename, which every task created before the rename keeps;
* hashed when present, null otherwise. Absent (undefined) on records
* captured before the field existed; comparisons skip a field the record
* predates, so old captures stay fresh until they are re-stamped.
*/
readonly graderGuidanceConsolidated?: string | null;
/**
* tests/holistic-rubric.md — the holistic rubric under its current
* filename (each task carries exactly one of this and the pre-rename
* name above). Hashed when present, null otherwise. Absent (undefined)
* on records captured before the field existed; comparisons skip a
* field the record predates. The task-checksum digest serializes fields
* in this order and appends new fields at the end (see
* scripts/lib/grader-run-checksums.ts).
*/
readonly holisticRubric?: string | null;
/**
* tests/atomic-rubric.yaml — the task's atomic rubric under its current
* filename. Hashed when present, null otherwise. Absent (undefined) on
* records captured before the field existed; comparisons skip a field
* the record predates.
*/
readonly atomicRubric?: string | null;
/**
* tests/rubrics.yaml — the atomic rubric under its pre-rename filename,
* which tasks converted before the rename keep. Hashed when present,
* null otherwise; absent (undefined) on records captured before the
* field existed.
*/
readonly rubricsYaml?: string | null;
/**
* tests/grader-context.md — the context document the rubric grader modes
* read beside the atomic rubric. Hashed when present, null otherwise;
* absent (undefined) on records captured before the field existed. Last
* in field order per the append-at-the-end digest rule above.
*/
readonly graderContext?: string | null;
};
}
export type TaskInputName = keyof TaskInputChecksums['inputs'];
/** Human-readable component names, used verbatim in staleness warnings. Frozen:
* its key set is the runtime source of truth for the task-input axes. */
export const INPUT_LABELS = Object.freeze({
prompt: 'prompt (instruction.md)',
graderGuidance: 'legacy-era grader guidance (tests/grader-guidance.md)',
graderGuidanceConsolidated: 'holistic rubric (tests/grader-guidance-consolidated.md)',
sessionJsonl: 'session snapshot (environment/session.jsonl)',
workspacePatch: 'workspace patch (environment/workspace.patch)',
gitref: 'gitref (task.toml commit)',
holisticRubric: 'holistic rubric (tests/holistic-rubric.md)',
atomicRubric: 'atomic rubric (tests/atomic-rubric.yaml)',
rubricsYaml: 'atomic rubric (tests/rubrics.yaml)',
graderContext: 'grader context (tests/grader-context.md)',
} as const satisfies Record<TaskInputName, string>);
/**
* The inputs that shape what the AGENT saw and did. Changing any of them means
* a captured reference run no longer reflects the task being packaged, and
* only re-running the agent can fix that. The holistic-rubric files are
* deliberately NOT in this set: editing the rubric stales the run's GRADE,
* not the run itself, and `scripts/harbor-regrade` re-derives grades without
* re-running the agent.
*/
export const REFERENCE_RUN_INPUTS = Object.freeze([
'prompt',
'sessionJsonl',
'workspacePatch',
'gitref',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs a detector report assesses — instruction.md plus whichever
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
* before the rename, plus the legacy-era plain-named file when a task
* authored on an earlier generation carries one), plus the atomic-rubric
* files the rubric detectors assess (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md). An absent file hashes to
* null on both sides and never diffs. Compared by content.
*/
export const DETECTOR_REPORT_INPUTS = Object.freeze([
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
'holisticRubric',
'atomicRubric',
'rubricsYaml',
'graderContext',
] as const satisfies readonly TaskInputName[]);
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
}
/**
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
* missing, unreadable, or has no commit line — a read failure downgrades to
* "absent" rather than crashing a capture or a validation sweep.
*
* Deliberately a regex, not a TOML parser: this module ships in the worker
* toolkit, where a new runtime dep would break packaging for every worker
* whose container predates the dep (npm install runs only on container
* create, and containers survive toolkit upgrades). build-workspace.sh reads
* the same key with the same grep-a-`commit`-line approach. The one `commit`
* key in a task.toml is `[metadata].commit`, so anchoring to the first
* `commit = "…"` line is exact in practice.
*/
function readGitref(taskDir: string): string | null {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return null;
try {
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
readFileSync(tomlPath, 'utf-8')
);
const commit = match?.[1] ?? match?.[2];
return commit && commit.length > 0 ? commit : null;
} catch {
return null;
}
}
/** Checksum the task inputs as they stand right now under `taskDir`. */
export function captureTaskInputs(
taskDir: string,
capturedBy?: TaskInputCaptureMethod
): TaskInputChecksums {
return Object.freeze({
version: INPUT_CHECKSUMS_VERSION,
capturedAt: new Date().toISOString(),
...(capturedBy ? { capturedBy } : {}),
inputs: Object.freeze({
prompt: sha256File(join(taskDir, 'instruction.md')),
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
gitref: readGitref(taskDir),
graderGuidanceConsolidated: sha256File(
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
),
holisticRubric: sha256File(join(taskDir, 'tests', 'holistic-rubric.md')),
atomicRubric: sha256File(join(taskDir, 'tests', 'atomic-rubric.yaml')),
rubricsYaml: sha256File(join(taskDir, 'tests', 'rubrics.yaml')),
graderContext: sha256File(join(taskDir, 'tests', 'grader-context.md')),
}),
});
}
/**
* Read a previously captured record. Returns null when the file is missing or
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
* future incompatible version) — callers treat null as "staleness unknowable",
* never as an error. No zod here: this module ships in the worker toolkit,
* whose dependency set stays minimal, so the guard is manual.
*/
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
if (!existsSync(filePath)) return null;
let parsed: unknown;
try {
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
} catch {
return null;
}
if (typeof parsed !== 'object' || parsed === null) return null;
const record = parsed as TaskInputChecksums;
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
const value = record.inputs[name];
// undefined = the record predates this input field; still a valid capture.
if (value !== undefined && value !== null && typeof value !== 'string') return null;
}
// An unrecognized capture method is dropped, not rejected: the field is
// provenance colour, and rejecting would flip the whole run to "unknowable".
const method: unknown = record.capturedBy;
const isKnown = TASK_INPUT_CAPTURE_METHODS.includes(method as TaskInputCaptureMethod);
if (method !== undefined && !isKnown) {
const { capturedBy: _dropped, ...rest } = record;
return rest;
}
return record;
}
/**
* Which of `names` changed between a recorded capture and the current state?
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
* whose value differs — including absent→present and present→absent flips. A
* field the recorded capture predates (the key is not in the record at all)
* is skipped: freshness on that axis is unknowable, and flagging every old
* record the moment a new axis ships would drown the real signal. (The
* task-checksum fold folds absence as 'null' instead — it only ever reads a
* fresh capture, which is total, so the two never disagree in practice.)
*/
export function diffTaskInputs(
recorded: TaskInputChecksums,
current: TaskInputChecksums,
names: readonly TaskInputName[]
): string[] {
return names
.filter((name) => name in recorded.inputs)
.filter((name) => recorded.inputs[name] !== current.inputs[name])
.map((name) => INPUT_LABELS[name]);
}

View File

@@ -0,0 +1,13 @@
/**
* Wrap a notice in a banner loud enough to survive a scrollback.
*
* Yellow only when stderr is a terminal, so piped logs stay clean.
*/
export function banner(message: string, headline: string): string {
const RULE = '#'.repeat(78);
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
return `${color[0]}${body}${color[1]}`;
}

View File

@@ -0,0 +1,431 @@
/**
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
*
* `environment/Dockerfile`, `tests/test.sh`, and
* `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
* are the same in every task: they decide how the
* trial runs and how the grade is produced. An edit makes a task's reference
* runs incomparable to every other task's, and the scores still look normal,
* so nothing downstream notices.
*
* A task is compared against itself as created. {@link writeManagedStamp} records
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
* creation, so a later mismatch is an edit made since. Tasks created before
* stamping have no record and fall back to matching the copies this toolkit
* ships — see {@link IntegrityStatus}.
*
* The toolkit appends to a task's Dockerfile itself (session staging, the
* reference-data corpus). Those blocks are wrapped in
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
* comparing, so they never read as edits.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { basename, join } from 'path';
import { banner } from './notice-banner.js';
/**
* Every status is advisory. Nothing here stops a trial or a submission: an author
* who changed one of these files did it because they didn't know we'd rather they
* didn't, and refusing to package their work punishes a misunderstanding. The job
* is to say so clearly, and to record it so a reviewer sees it too.
*
* `ok` — identical to a copy this toolkit ships, or unchanged since the
* task was created.
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
* copy since. Nobody's mistake; it does mean this task's runs
* aren't directly comparable to one built today.
* `modified` — matches neither its baseline nor anything shipped: an edit.
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
* and an older release are indistinguishable.
* `missing` — the task doesn't have the file.
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
* been selected yet.
*/
export type IntegrityStatus =
| 'ok'
| 'outdated'
| 'modified'
| 'unverifiable'
| 'missing'
| 'placeholder';
export interface FileVerdict {
/** Task-relative path, e.g. `environment/Dockerfile`. */
taskPath: string;
status: IntegrityStatus;
/** Command that restores the managed version, on `modified` / `unverifiable`. */
restore?: string;
}
export interface IntegrityReport {
/**
* False when this isn't a worker toolkit, or when the task was authored on
* a different toolkit generation (see {@link isTaskFromThisToolkitGeneration})
* — callers should skip silently.
*/
checked: boolean;
files: FileVerdict[];
/** Looks like an edit: matches neither a baseline nor anything shipped. */
modified: FileVerdict[];
/** Can't be told apart from an older release. */
unverifiable: FileVerdict[];
/** Unchanged, but a newer copy has shipped since. */
outdated: FileVerdict[];
}
interface ManagedFile {
taskPath: string;
/** Matches the candidate pristine filenames under `task-shared/`. */
baselinePattern: RegExp;
}
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
const MANAGED_FILES: ManagedFile[] = [
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
{
taskPath: 'tests/grader-system-prompt-consolidated.md',
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
},
];
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
/**
* Line shapes from toolkit releases that predate the sentinels. Deliberately
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
* lines" rule an edit could hide behind.
*/
const LEGACY_MANAGED_LINES: RegExp[] = [
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
];
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
/** Per-task stamp of the managed files as created. Lives in the task directory. */
export const STAMP_FILENAME = '.toolkit-managed.json';
interface ManagedStamp {
version: number;
stampedAt: string;
/** taskPath → sha256 of the stripped content. */
files: Record<string, string>;
}
/**
* Remove toolkit-appended content so only author-authored differences remain.
* Trailing blank lines go too — an editor adding or trimming a final newline is
* not something to fail a trial over.
*/
export function stripManagedBlocks(content: string): string {
const out: string[] = [];
let inBlock = false;
// Normalize CRLF before anything else: a Windows editor or a checkout with
// core.autocrlf rewrites every line ending, and that must not read as an edit.
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
if (!inBlock && SENTINEL_OPEN.test(line)) {
inBlock = true;
continue;
}
if (inBlock) {
if (SENTINEL_CLOSE.test(line)) inBlock = false;
continue;
}
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
out.push(line);
}
return out.join('\n').replace(/\s+$/, '');
}
export function sha256(content: string): string {
return createHash('sha256').update(content).digest('hex');
}
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
if (!existsSync(sharedDir)) return [];
return readdirSync(sharedDir)
.filter((f) => pattern.test(f))
.sort();
}
/**
* Record the managed files, so later edits are detectable. Call at task creation
* and after a managed file is first put in place.
*
* A file earns a baseline only by matching a copy this toolkit ships, and an
* entry already recorded is never rewritten. Together those mean a stamp can
* only ever describe a pristine file: re-running this can't turn an author's
* edit into the new baseline, and a file dropped in later (the polyglot
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
* baseline once it's in place.
*
* Returns true if anything was recorded.
*/
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
const sharedDir = join(toolkitRoot, 'task-shared');
const existing = readStamp(taskDir);
const files: Record<string, string> = { ...(existing?.files ?? {}) };
let added = false;
for (const managed of MANAGED_FILES) {
if (files[managed.taskPath]) continue;
const p = join(taskDir, managed.taskPath);
if (!existsSync(p)) continue;
const raw = readFileSync(p, 'utf-8');
// Not a baseline: the author still has to drop in their member's base image.
if (raw.includes(PLACEHOLDER_MARKER)) continue;
const stripped = stripManagedBlocks(raw);
if (!matchesShipped(sharedDir, managed, stripped)) continue;
files[managed.taskPath] = sha256(stripped);
added = true;
}
if (!added) return false;
const stamp: ManagedStamp = {
version: 1,
stampedAt: new Date().toISOString(),
files,
};
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
return true;
}
function readStamp(taskDir: string): ManagedStamp | null {
const stampPath = join(taskDir, STAMP_FILENAME);
if (!existsSync(stampPath)) return null;
try {
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
return null;
}
}
/**
* How to restore a managed file, or undefined when this toolkit ships no copy to
* restore from. Only the Dockerfile can have several candidates (one per member).
*/
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
const dest = `harbor-tasks/${slug}/${taskPath}`;
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
if (candidates.length > 1) {
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
}
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
// author to copy a Dockerfile over their grader prompt.
return undefined;
}
/** Render a restore line, saying so plainly when there is nothing to restore from. */
function restoreLine(f: FileVerdict): string {
return f.restore
? ` ${f.restore}`
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
}
/** Does this content match a pristine copy the toolkit ships? */
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
return baselineCandidates(sharedDir, managed.baselinePattern).some(
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
);
}
/**
* Was this task created by this toolkit generation? Task creation (the packed
* scaffold's task.toml and snapshot-to-task) writes `[metadata].toolkit_version`;
* a task directory without the key was authored on a different toolkit
* generation and grades with the assets frozen in its own tests/ directory, so
* comparing those against this toolkit's copies would report drift that is not
* an edit. Presence-based on purpose: wall-clock stamps cannot separate the
* generations, because tasks from an earlier generation are completed after
* later kits ship.
*
* A regex rather than a TOML parser, for the same shipped-dependency reason as
* input-checksums.ts readGitref: the one `toolkit_version` key in a task.toml
* is `[metadata].toolkit_version`.
*/
export function isTaskFromThisToolkitGeneration(taskDir: string): boolean {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return false;
try {
return /^\s*toolkit_version\s*=/m.test(readFileSync(tomlPath, 'utf-8'));
} catch {
return false;
}
}
/**
* Compare a task's managed files against its creation-time stamp.
*
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
*/
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
const sharedDir = join(toolkitRoot, 'task-shared');
// Without task-shared/ there is nothing to compare against; report "not
// checked" so callers no-op rather than reporting three phantom failures.
if (!existsSync(sharedDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
// A task authored on a different toolkit generation grades with the assets
// frozen in its own tests/ directory. Comparing those against this toolkit's
// copies would report drift that is not an edit — and the printed remedy
// (restore the current copy) would change how that task grades. Skip it.
if (!isTaskFromThisToolkitGeneration(taskDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
const slug = basename(taskDir);
const stamp = readStamp(taskDir);
const files: FileVerdict[] = [];
for (const managed of MANAGED_FILES) {
const taskFile = join(taskDir, managed.taskPath);
if (!existsSync(taskFile)) {
files.push({ taskPath: managed.taskPath, status: 'missing' });
continue;
}
const raw = readFileSync(taskFile, 'utf-8');
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
const restore = restoreCommand(managed.taskPath, candidates, slug);
const stripped = stripManagedBlocks(raw);
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
// cannot be an author edit, whatever the stamp says — and asking the stamp first
// is what used to make restoring the current copy (which is exactly what we tell
// authors to do) look like an edit, with no way out.
if (matchesShipped(sharedDir, managed, stripped)) {
files.push({ taskPath: managed.taskPath, status: 'ok' });
continue;
}
const expected = stamp?.files[managed.taskPath];
if (expected) {
// Matches its baseline but nothing shipped: untouched by the author, and the
// toolkit has moved on since. Worth saying, nobody's fault.
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
files.push({ taskPath: managed.taskPath, status, restore });
continue;
}
// Checked after the stamp so that adding this marker to a file that HAS a
// baseline can't exempt it from the comparison.
if (raw.includes(PLACEHOLDER_MARKER)) {
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
continue;
}
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
}
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
unverifiable: files.filter((f) => f.status === 'unverifiable'),
outdated: files.filter((f) => f.status === 'outdated'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal. An author who changed one of these files
* almost always did it to get unstuck, not knowing we'd rather they told us — so
* this explains what it means for their task and what restoring would do, and then
* lets them get on with it.
*/
export function formatIntegrityReport(report: IntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These files look edited since this task was created, and the toolkit manages',
'them — they set up how the trial runs and how the grade is produced, so they',
"have to be identical across every task. Yours aren't, which makes this task's",
"runs hard to compare with everyone else's:",
'',
...report.modified.map((f) => ` ${f.taskPath}`),
'',
'Restoring the shipped version puts that right:',
...report.modified.map(restoreLine),
'',
'If you changed one to work around a problem — a missing package, a grader that',
"wouldn't run — please tell us about the problem instead. It almost certainly",
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.outdated.length > 0) {
sections.push(
[
'These files are unchanged, but the toolkit has shipped newer copies since this',
'task was created:',
'',
...report.outdated.map((f) => ` ${f.taskPath}`),
'',
"You haven't done anything wrong. It does mean this task was run and graded with",
"older versions than a task built today, so its scores aren't directly",
'comparable. To line them up, restore the current copies and re-run your trials:',
...report.outdated.map(restoreLine),
].join('\n')
);
}
if (report.unverifiable.length > 0) {
sections.push(
[
"These files don't match the copies this toolkit ships, and this task has no",
'record of what they looked like when it was created:',
'',
...report.unverifiable.map((f) => ` ${f.taskPath}`),
'',
'Two things look like this and we cannot tell them apart: a task created on an',
'earlier toolkit release (nothing to fix, though its scores are not directly',
'comparable to a task built today), or a file that was edited. Either way,',
'restoring the current copy and re-running your trials is what makes this task',
"comparable to everyone else's:",
...report.unverifiable.map(restoreLine),
].join('\n')
);
}
return sections.join('\n\n');
}
/**
* Wrap a report in a banner loud enough to survive a scrollback.
*
* Nothing blocks any more, so this notice is the entire mechanism — and an
* unframed paragraph among build output is one a reasonable person scrolls past.
*/
export function bannerize(message: string, report: IntegrityReport): string {
return banner(
message,
report.modified.length > 0
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!'
);
}

View File

@@ -0,0 +1,217 @@
/**
* toolkit-script-integrity.ts — detect edits to the toolkit's own scripts.
*
* Sibling of task-infra-integrity.ts, which covers a task's managed files. This
* covers `scripts/`. The scripts never ship with a task, so an edit can't reach
* the delivered workspace — but their OUTPUT does: `build-workspace.sh` alone
* stages `tests/test-commands.sh` (the deterministic checks behind the
* correctness score), writes the Dockerfile's toolkit-managed blocks, and
* records the managed stamp and input checksums. Nothing downstream re-derives
* those, and the reference runs can't be re-derived at all.
*
* The baseline is a manifest written at package time ({@link writeScriptManifest}),
* so it ships in the same zip as the scripts it describes. That removes the
* ambiguity a task's managed files have: there is no "created on an older
* release" case to tell apart, so a hash mismatch is an edit. Files absent from
* the manifest are ignored, which keeps a worker's own helper script — or a
* `__pycache__` left by a harbor run — from ever being reported.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { join, relative } from 'path';
import { banner } from './notice-banner.js';
/** Manifest of the shipped `scripts/` tree. Lives at the toolkit root. */
export const SCRIPT_MANIFEST_FILENAME = '.toolkit-scripts.json';
const MANIFEST_VERSION = 1;
/** Runtime droppings, never part of the shipped tree. */
const IGNORED_DIRS = new Set(['__pycache__', 'node_modules', '.git']);
const IGNORED_FILES = /\.(pyc|pyo)$/;
/**
* `modified` — content differs from what shipped: an edit.
* `missing` — shipped, but no longer on disk.
* `ok` — unchanged.
*/
export type ScriptStatus = 'ok' | 'modified' | 'missing';
export interface ScriptVerdict {
/** Toolkit-relative path, e.g. `scripts/build-workspace.sh`. */
path: string;
status: ScriptStatus;
}
export interface ScriptIntegrityReport {
/** False when no manifest ships — callers should skip silently. */
checked: boolean;
files: ScriptVerdict[];
modified: ScriptVerdict[];
missing: ScriptVerdict[];
}
interface ScriptManifest {
version: number;
generatedAt: string;
/** Toolkit-relative path → sha256 of the normalized content. */
files: Record<string, string>;
}
/**
* Line endings and trailing whitespace are normalized away: a Windows editor, a
* checkout with core.autocrlf, or a formatter trimming a final newline must not
* read as an edit.
*/
function hashContent(content: string): string {
return createHash('sha256')
.update(content.replace(/\r\n/g, '\n').replace(/\s+$/, ''))
.digest('hex');
}
/** Every shipped file under `scripts/`, as toolkit-relative paths. */
function walkScripts(dir: string, toolkitRoot: string): string[] {
if (!existsSync(dir)) return [];
const out: string[] = [];
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) =>
a.name.localeCompare(b.name)
)) {
const abs = join(dir, entry.name);
if (entry.isDirectory()) {
if (!IGNORED_DIRS.has(entry.name)) out.push(...walkScripts(abs, toolkitRoot));
continue;
}
if (!entry.isFile() || IGNORED_FILES.test(entry.name)) continue;
out.push(relative(toolkitRoot, abs));
}
return out;
}
/**
* Record the shipped `scripts/` tree. Call at package time, once the tree is
* fully staged — anything written to `scripts/` afterwards reads as an edit.
*
* Returns the number of files recorded.
*/
export function writeScriptManifest(toolkitRoot: string): number {
const files: Record<string, string> = {};
for (const rel of walkScripts(join(toolkitRoot, 'scripts'), toolkitRoot)) {
files[rel] = hashContent(readFileSync(join(toolkitRoot, rel), 'utf-8'));
}
const manifest: ScriptManifest = {
version: MANIFEST_VERSION,
generatedAt: new Date().toISOString(),
files,
};
writeFileSync(
join(toolkitRoot, SCRIPT_MANIFEST_FILENAME),
`${JSON.stringify(manifest, null, 2)}\n`
);
return Object.keys(files).length;
}
function readManifest(toolkitRoot: string): ScriptManifest | null {
const p = join(toolkitRoot, SCRIPT_MANIFEST_FILENAME);
if (!existsSync(p)) return null;
try {
const parsed = JSON.parse(readFileSync(p, 'utf-8')) as ScriptManifest;
if (parsed?.version !== MANIFEST_VERSION) return null;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// A corrupt manifest is treated as no manifest rather than blocking a trial.
return null;
}
}
/**
* Compare the toolkit's `scripts/` tree against the manifest it shipped with.
*
* @param toolkitRoot Absolute path to the toolkit root (holds `scripts/`).
*/
export function checkToolkitScriptIntegrity(toolkitRoot: string): ScriptIntegrityReport {
const manifest = readManifest(toolkitRoot);
if (!manifest) return { checked: false, files: [], modified: [], missing: [] };
const files: ScriptVerdict[] = Object.entries(manifest.files).map(([path, expected]) => {
const abs = join(toolkitRoot, path);
if (!existsSync(abs)) return { path, status: 'missing' as const };
const status = hashContent(readFileSync(abs, 'utf-8')) === expected ? 'ok' : 'modified';
return { path, status };
});
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
missing: files.filter((f) => f.status === 'missing'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal, for the same reason the managed-file
* notice isn't: an author who changed one of these did it to get unstuck, and the
* fix they needed almost certainly belongs in the toolkit rather than in their copy.
*/
export function formatScriptIntegrityReport(report: ScriptIntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These toolkit scripts look edited:',
'',
...report.modified.map((f) => ` ${f.path}`),
'',
"They aren't part of any task, so an edit is easy to miss — but what they write",
'is. Building a task stages its deterministic checks, fills in parts of its',
'Dockerfile, and records the checksums a reviewer reads; a script that does any of',
'that differently produces a task that looks normal and behaves differently from',
'every other one.',
'',
'Re-extracting the toolkit zip over your copy restores them. Your tasks, snapshots',
'and reference runs are untouched by that.',
'',
'If you changed one to work around a problem — a build that would not run, a',
'missing dependency — please tell us about the problem instead. It almost',
'certainly affects other authors too, and the fix belongs in the toolkit.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.missing.length > 0) {
sections.push(
[
'These toolkit scripts shipped with this release but are no longer here:',
'',
...report.missing.map((f) => ` ${f.path}`),
'',
'Something that depends on one will fail partway through rather than up front.',
'Re-extract the toolkit zip over your copy to put them back.',
].join('\n')
);
}
return sections.join('\n\n');
}
/** The full notice, bannered and ready to write to stderr, or '' if all is well. */
export function scriptIntegrityNotice(report: ScriptIntegrityReport): string {
const message = formatScriptIntegrityReport(report);
if (!message) return '';
const headline =
report.modified.length > 0
? '!! TOOLKIT SCRIPTS LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT SCRIPTS ARE MISSING — PLEASE READ !!';
return banner(message, headline);
}

View File

@@ -0,0 +1,304 @@
/**
* Tests for tree-permissions.ts.
*
* The load-bearing case is the one from the field report: a directory that came
* across without its search bit makes `tar` fail with `Cannot stat` on the files
* *inside* it, so the repair has to fix directory modes, not just ownership.
* These tests run unprivileged, so they exercise the mode axis for real and the
* ownership axis only as far as an unprivileged process can (target resolution +
* graceful EPERM), which is the same shape CI runs in. One case needs root and
* skips otherwise; the rest hold under either uid, which is why the fixtures that
* must look human-owned say so with `ownedByHuman` instead of relying on the
* caller's uid.
*/
import assert from 'node:assert/strict';
import {
chmodSync,
chownSync,
mkdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { test } from 'node:test';
import {
didRepair,
manualRepairHint,
normalizeTreePermissions,
resolveWorkspaceOwner,
} from './tree-permissions';
function scratch(name: string): string {
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
rmSync(dir, { recursive: true, force: true });
mkdirSync(dir, { recursive: true });
return dir;
}
const RUNNING_AS_ROOT = process.getuid?.() === 0;
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
function ownedByHuman(path: string): string {
chownSync(path, HUMAN_UID, HUMAN_GID);
return path;
}
test('restores the search bit on a directory that lost it', () => {
const root = scratch('searchbit');
const models = join(root, 'agent-output', 'app', 'models');
mkdirSync(models, { recursive: true });
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
// r-- : readdir works, so tar can NAME the file, but stat is refused.
chmodSync(models, 0o400);
const report = normalizeTreePermissions(root);
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
assert.ok(report.modeFixed.some((p) => p === models));
assert.ok(didRepair(report));
rmSync(root, { recursive: true, force: true });
});
test('recurses into a directory it had to widen first', () => {
const root = scratch('recurse');
const inner = join(root, 'locked', 'deeper');
mkdirSync(inner, { recursive: true });
const leaf = join(inner, 'leaf.rb');
writeFileSync(leaf, 'x\n');
chmodSync(leaf, 0o000);
chmodSync(inner, 0o400);
chmodSync(join(root, 'locked'), 0o400);
const report = normalizeTreePermissions(root);
// Only reachable if the walk widened each parent before descending.
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
assert.ok(report.modeFixed.includes(leaf));
rmSync(root, { recursive: true, force: true });
});
test('leaves already-correct trees untouched', () => {
const root = scratch('noop');
mkdirSync(join(root, 'sub'), { recursive: true });
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
const report = normalizeTreePermissions(root);
assert.deepEqual(report.modeFixed, [], 'no mode changes');
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
rmSync(root, { recursive: true, force: true });
});
test('does not widen group/other beyond what was already there', () => {
const root = ownedByHuman(scratch('narrow'));
const f = join(root, 'secret.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o000);
normalizeTreePermissions(root, { ownerRef: root });
const mode = statSync(f).mode & 0o777;
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
rmSync(root, { recursive: true, force: true });
});
test('ignores symlinks rather than following them out of the tree', () => {
const root = scratch('symlink');
const outside = scratch('symlink-outside');
const victim = join(outside, 'victim.txt');
writeFileSync(victim, 'x\n');
chmodSync(victim, 0o000);
symlinkSync(outside, join(root, 'link'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
rmSync(outside, { recursive: true, force: true });
});
test('never throws on a missing root, and reports it', () => {
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
assert.equal(report.failures.length, 1);
assert.equal(report.failures[0].reason, 'ENOENT');
});
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
const root = scratch('owner');
const owner = resolveWorkspaceOwner(root);
assert.ok(owner, 'resolved');
const st = statSync(root);
assert.equal(owner.uid, st.uid);
assert.equal(owner.gid, st.gid);
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
rmSync(root, { recursive: true, force: true });
});
test('never chowns TO root, even when the owner ref is root-owned', () => {
// The regression this guards: workspace root owned by root (unzipped with
// sudo) while the task files are correctly owned by the human. Chowning to the
// ref's owner would inflict the very lockout this module prevents. `/` is
// root-owned on every platform we run on, so it's a stable stand-in.
const root = scratch('root-ref');
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
const beforeUid = statSync(f).uid;
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(report.target?.uid, 0, 'resolved a root target');
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
assert.deepEqual(report.failures, [], 'and did not fail trying');
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
rmSync(root, { recursive: true, force: true });
});
test('still normalizes modes when the chown target is root', () => {
const root = scratch('root-ref-modes');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
writeFileSync(join(sub, 'f.txt'), 'x\n');
chmodSync(sub, 0o400);
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
assert.ok(report.modeFixed.includes(sub));
rmSync(root, { recursive: true, force: true });
});
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
// The complement of the case below: we declined to chown, but these entries are
// already the human's, so owner bits reach them and nothing should be widened.
const root = ownedByHuman(scratch('root-ref-narrow'));
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o600);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
rmSync(root, { recursive: true, force: true });
});
test(
'grants read+search to group and other on files stranded root-owned',
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
() => {
// The worker authoring container: root process, root-owned workspace. The chown
// is declined, so owner bits land on root and the human — a different uid in
// Explore and on a WSL host — is still locked out of a --w------- capture.
const root = scratch('stranded');
const sub = join(root, 'agent-output');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'answer.md');
writeFileSync(f, 'x\n');
chmodSync(f, 0o200);
chmodSync(sub, 0o300);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(
statSync(f).mode & 0o777,
0o644,
'file readable by everyone, writable by none but root'
);
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
}
);
test('walks a tree as deep as the filesystem allows', () => {
const root = scratch('deep');
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
// call-stack limit. So this isn't a stack test; it just pins that a deep,
// narrow tree walks cleanly end to end.
let path = root;
for (let i = 0; i < 250; i++) {
path = join(path, `d${i}`);
}
mkdirSync(path, { recursive: true });
writeFileSync(join(path, 'leaf.txt'), 'x\n');
chmodSync(join(path, 'leaf.txt'), 0o000);
const report = normalizeTreePermissions(root);
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
rmSync(root, { recursive: true, force: true });
});
test('a failure in one subtree does not abandon the rest', () => {
const root = scratch('partial');
const good = join(root, 'good');
mkdirSync(good, { recursive: true });
const goodFile = join(good, 'f.txt');
writeFileSync(goodFile, 'x\n');
chmodSync(goodFile, 0o000);
// A dangling symlink and a vanished path both produce per-entry trouble.
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
assert.ok(report.modeFixed.includes(goodFile));
rmSync(root, { recursive: true, force: true });
});
test('reports rather than throws when the root is a file, not a directory', () => {
const root = scratch('file-root');
const f = join(root, 'lonely.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
const report = normalizeTreePermissions(f);
assert.equal(statSync(f).mode & 0o600, 0o600);
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
});
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
const root = scratch('killswitch');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'f.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
chmodSync(sub, 0o400);
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
try {
const report = normalizeTreePermissions(root);
assert.equal(report.skipped, true);
assert.deepEqual(report.modeFixed, []);
assert.deepEqual(report.ownerFixed, []);
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
} finally {
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
}
chmodSync(sub, 0o700);
rmSync(root, { recursive: true, force: true });
});
test('manual hint repairs both axes, ownership first', () => {
const hint = manualRepairHint('harbor-tasks/my-slug');
assert.match(hint, /chown -R/);
assert.match(hint, /chmod -R u\+rwX/);
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
});

View File

@@ -0,0 +1,126 @@
/**
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
*
* Files captured from a task run can arrive owned by another user, or with a
* directory missing the permission needed to walk into it. Packaging then fails
* with `Cannot stat: Permission denied`. This repairs both.
*
* Grants owner rwX only, never group or other. Never throws, and never hands
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
*/
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
import { join } from 'path';
export interface NormalizeReport {
/** Paths whose owner was changed. */
ownerFixed: string[];
/** Paths whose mode gained owner rwX. */
modeFixed: string[];
/** Paths we wanted to change but could not, with the errno. */
failures: { path: string; reason: string }[];
/** Resolved target owner, or null if it couldn't be determined. */
target: { uid: number; gid: number } | null;
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
skipped?: boolean;
}
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
try {
const st = statSync(ownerRef);
return { uid: st.uid, gid: st.gid };
} catch {
return null;
}
}
/**
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
*
* `stranded` means the file stays root-owned because we have no non-root owner to
* give it to. Owner bits then help nobody — whoever has to read it is a different
* user — so read and search are granted more widely. Never write, never +x on files.
*/
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
const owner = isDir ? 0o700 : 0o600;
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
}
/**
* Give every entry under `root` to the workspace owner and make sure that owner
* can read and traverse it. Symlinks are skipped. Repairs what it can and
* reports what it couldn't; it never throws and never blocks its caller.
*/
export function normalizeTreePermissions(
root: string,
options: { ownerRef?: string } = {}
): NormalizeReport {
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
}
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
// Never hand files to root — that would lock the owner out rather than help.
const chownTarget = target && target.uid !== 0 ? target : null;
try {
const stack: string[] = [root];
while (stack.length > 0) {
const path = stack.pop() as string;
let st;
try {
st = lstatSync(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
continue;
}
if (st.isSymbolicLink()) continue;
const isDir = st.isDirectory();
// Mode first: a directory we can't search is one we can't descend into.
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
if (wanted !== st.mode) {
try {
chmodSync(path, wanted);
report.modeFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
}
}
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
try {
chownSync(path, chownTarget.uid, chownTarget.gid);
report.ownerFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
}
}
if (!isDir) continue;
try {
for (const entry of readdirSync(path)) stack.push(join(path, entry));
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
}
}
} catch (err) {
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
}
return report;
}
/** True when something was actually repaired. */
export function didRepair(report: NormalizeReport): boolean {
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
}
/** The command to run on your host if we couldn't fix it ourselves. */
export function manualRepairHint(path: string): string {
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
}

View File

@@ -0,0 +1,80 @@
/**
* record-detector-inputs.ts — stamp detector report(s) with the checksums of
* the task inputs they assessed.
*
* Run this right after a detector skill writes (or rewrites)
* harbor-tasks/<slug>/detectors/<detector-name>.md. It records a sha256
* capture of the task inputs next to the report, as
* detectors/<detector-name>.inputs.json, so submit-task.ts can tell by
* content — not by file timestamp — whether the report still matches the
* task being packaged.
*
* Usage:
* npx tsx scripts/record-detector-inputs.ts <task-slug> <detector-name> [detector-name...]
* npx tsx scripts/record-detector-inputs.ts my-cool-task detector-rubric-clarity
*/
import { existsSync, mkdirSync, writeFileSync } from 'fs';
import { join } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { captureTaskInputs } from './lib/input-checksums';
const argv = yargs(hideBin(process.argv))
.usage(
'$0 <slug> <detectors...>',
'Record the task-input checksums a detector report assessed',
(y) =>
y
.positional('slug', { type: 'string', demandOption: true, describe: 'Task slug' })
.positional('detectors', {
type: 'string',
array: true,
demandOption: true,
describe: 'Detector name(s), e.g. detector-rubric-clarity',
})
)
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.help()
.parseSync();
const log = pino(
{ name: 'record-detector-inputs', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
const slug = argv.slug as string;
const detectors = argv.detectors as string[];
const taskDir = join(process.cwd(), 'harbor-tasks', slug);
if (!existsSync(taskDir)) {
log.fatal({ taskDir }, 'Task directory not found');
process.exit(1);
}
// One capture serves every report stamped in this invocation — they all
// assessed the same on-disk revision of the task.
const capture = captureTaskInputs(taskDir, 'stamp');
const detectorsDir = join(taskDir, 'detectors');
mkdirSync(detectorsDir, { recursive: true });
for (const name of detectors) {
const report = join(detectorsDir, `${name}.md`);
if (!existsSync(report)) {
// Stamp anyway — skills sometimes stamp before the final write lands —
// but say so, since a stamp with no report usually means a typo'd name.
log.warn({ report: `detectors/${name}.md` }, 'No report found for this detector name');
}
const stampPath = join(detectorsDir, `${name}.inputs.json`);
writeFileSync(stampPath, JSON.stringify(capture, null, 2) + '\n');
log.info({ stamp: `detectors/${name}.inputs.json` }, 'Recorded task-input checksums');
}

View File

@@ -0,0 +1,166 @@
"""Pure (no-harbor) helpers for inspecting a captured reference run.
Kept separate from ``replay_agent.py`` (which imports ``harbor``) so this logic
can be unit-tested with the plain devcontainer Python and shipped in the worker
toolkit alongside the replay agent.
The one safety-critical helper here is :func:`captured_mutating_tools`: it tells
the replay agent whether a run with no ``agent-output/`` is a harmless advisory
run (the agent only read + answered in chat) or a genuine capture loss (the
agent edited files but they weren't preserved). The replay agent grades the
former from the captured transcript and refuses the latter.
Structured edit tools (``Write``/``Edit``/``MultiEdit``/``NotebookEdit``) are
obvious. ``Bash`` is the subtle one: a shell call can mutate the workspace
(``rm``, ``mv``, ``sed -i``, ``echo … > f`` …) just as easily as it can read it.
So a ``Bash`` call is treated as **potentially mutating unless the command is
verifiably read-only** (:func:`bash_mutates`) — the safe direction: an unknown
command counts as a mutation, so we never silently grade a run that lost edits.
"""
from __future__ import annotations
import json
import re
from pathlib import Path
# Structured tools that always mutate the workspace.
MUTATING_TOOLS = frozenset({"Write", "Edit", "MultiEdit", "NotebookEdit"})
# Base commands that only read (or touch non-workspace state like cwd). Anything
# NOT here — or any file-writing redirection, or `sed -i`, or a non-read-only git
# subcommand — is treated as potentially mutating.
_READONLY_BASH = frozenset({
"ls", "cat", "head", "tail", "grep", "egrep", "fgrep", "rg", "ag", "find",
"fd", "wc", "echo", "printf", "file", "stat", "pwd", "tree", "sort", "uniq",
"cut", "tr", "awk", "jq", "yq", "less", "more", "diff", "cmp", "basename",
"dirname", "realpath", "readlink", "true", "false", "test", "[", "date",
"env", "printenv", "which", "type", "command", "column", "nl", "od", "xxd",
"hexdump", "comm", "paste", "fold", "expand", "tac", "du", "df", "seq",
"sleep", ":", "cd", "pushd", "popd", "dirs", "whoami", "hostname", "uname",
"id", "cksum", "md5sum", "sha1sum", "sha256sum", "strings", "wc",
})
# git subcommands that don't write the repo/workspace.
_READONLY_GIT_SUB = frozenset({
"log", "diff", "status", "show", "blame", "grep", "ls-files", "ls-tree",
"cat-file", "rev-parse", "describe", "shortlog", "reflog", "rev-list",
"for-each-ref", "name-rev", "symbolic-ref", "whatchanged", "var", "help",
"show-ref", "merge-base", "cherry", "count-objects", "verify-pack",
})
# fd-dups (2>&1, >&2, 1>&-) and /dev/null sinks are harmless; strip them before
# looking for a real file-writing redirection.
_HARMLESS_REDIR = re.compile(r"[0-9&]*>>?\s*(?:&\s*[0-9-]+|/dev/null)")
# Split a command line into segments on shell separators + substitutions.
_SEG_SPLIT = re.compile(r"\|\||&&|[|;&\n]|\$\(|`")
_ASSIGN = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*=")
def bash_mutates(command: str) -> bool:
"""Heuristic: does this shell command potentially write the workspace?
Conservative by design — errs toward True (an unrecognized command, a
file-writing redirect, `sed -i`, or a non-read-only git subcommand all count
as mutating). Read-only exploration (ls/cat/grep/find/wc/… piped together,
with `2>/dev/null` / `>/dev/null`) returns False."""
if not command or not command.strip():
return False
# Drop fd-dups (2>&1) + /dev/null sinks up front, so they neither look like a
# file write nor split a segment on their `&` (2>&1 → bogus "1" command).
stripped = _HARMLESS_REDIR.sub(" ", command)
# 1. Any remaining redirection now writes a real file.
if ">" in stripped:
return True
# 2. The leading command of every segment must be read-only.
for seg in _SEG_SPLIT.split(stripped):
toks = seg.split()
idx = 0
while idx < len(toks) and _ASSIGN.match(toks[idx]): # skip VAR=val prefixes
idx += 1
if idx >= len(toks):
continue
cmd = toks[idx].rsplit("/", 1)[-1]
rest = toks[idx + 1:]
if cmd == "sed" and any(t == "-i" or t.startswith("-i") for t in rest):
return True
if cmd == "git":
sub = next((t for t in rest if not t.startswith("-")), "")
if sub and sub not in _READONLY_GIT_SUB:
return True
continue
if cmd and cmd not in _READONLY_BASH:
return True
return False
def _bash_command(call_args) -> str:
if isinstance(call_args, dict):
return str(call_args.get("command", "") or "")
return ""
def _scan_trajectory(trajectory_path: Path) -> set[str]:
"""Mutating tool names in an ATIF agent/trajectory.json
(steps[].tool_calls[].function_name; Bash inspected by command)."""
found: set[str] = set()
if not trajectory_path.exists():
return found
try:
data = json.loads(trajectory_path.read_text())
except (json.JSONDecodeError, OSError):
return found
for step in data.get("steps", []):
for call in step.get("tool_calls") or []:
name = call.get("function_name")
if name in MUTATING_TOOLS:
found.add(name)
elif name == "Bash" and bash_mutates(_bash_command(call.get("arguments"))):
found.add("Bash")
return found
def _scan_stream_json(stream_path: Path) -> set[str]:
"""Mutating tool names in a raw stream-json claude-code.txt
(one JSON object per line, message.content[].tool_use; Bash by input)."""
found: set[str] = set()
if not stream_path.exists():
return found
try:
lines = stream_path.read_text(errors="ignore").splitlines()
except OSError:
return found
for line in lines:
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except json.JSONDecodeError:
continue
message = obj.get("message") if isinstance(obj, dict) else None
content = message.get("content") if isinstance(message, dict) else None
if not isinstance(content, list):
continue
for block in content:
if not (isinstance(block, dict) and block.get("type") == "tool_use"):
continue
name = block.get("name")
if name in MUTATING_TOOLS:
found.add(name)
elif name == "Bash" and bash_mutates(_bash_command(block.get("input"))):
found.add("Bash")
return found
def captured_mutating_tools(reference_run_dir: Path | str) -> set[str]:
"""Return the file-mutating tool names found in a captured run's transcript.
Checks the ATIF ``agent/trajectory.json`` first, then falls back to the raw
stream-json ``agent/claude-code.txt``, so a lossy/partial trajectory can't
hide a real edit. ``Bash`` is included only when its command isn't verifiably
read-only (see :func:`bash_mutates`). An empty result means the agent made no
workspace edits — i.e. a missing ``agent-output/`` is an advisory no-op.
"""
ref = Path(reference_run_dir)
return _scan_trajectory(ref / "agent" / "trajectory.json") | _scan_stream_json(
ref / "agent" / "claude-code.txt"
)

View File

@@ -0,0 +1,37 @@
#!/bin/bash
# Rewrite the auth FILES harnesses read their key from — and the base URL beside them —
# off the live .env, then exec "$@".
#
# codex reads its key from ${CODEX_HOME:-$HOME/.codex}/auth.json, which container-create
# wrote once from the .env of that moment — so a key rotated afterwards never reached it
# and needed a rebuild. claude needs none of this: it has an apiKeyHelper that re-reads
# .env per request. Interactive launches route through here so each one re-derives first.
#
# The base URL never rotates, so the case that matters is the one where container-create
# could not derive it at all (no .env yet) and wrote no config: the key then refreshes
# fine while codex still has no proxy URL and talks to the provider directly.
#
# Trials are unaffected either way: harbor-run re-derives OPENAI_API_KEY per invocation
# and harbor's codex agent authenticates the sandbox from that env var, not from this file.
set -uo pipefail
_scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# Subshell, and every failure swallowed: a refresh that cannot run must never stop the
# agent from starting. The auth file already on disk is the PREVIOUS key, not nothing, so
# failing open leaves the worker exactly where they were before this wrapper existed.
(
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$_scripts_dir" . "$_scripts_dir/lib/harness-credentials.sh" || exit 0
harness_setup_credentials
harness_write_auth
harness_refresh_config_keys
) >/dev/null 2>&1 || true
# No args is a valid call: refresh only, for a lifecycle hook.
[ "$#" -gt 0 ] || exit 0
exec "$@"

View File

@@ -0,0 +1,201 @@
"""
Harbor agent adapter that re-grades an existing reference run by replaying
its captured workspace state — no model calls, no agent work.
Reads a `reference_run_dir` pointing at a `reference-runs/<id>/` directory
captured during a prior real trial. Inside that dir:
agent/trajectory.json — the grader reads this via the symlink
/tmp/outputs/task_transcript.txt → /logs/agent/trajectory.json
agent-output/ — files the agent created or modified (captured by
tests/test.sh after the agent ran)
agent-output/_HARBOR_DELETIONS.txt
— list of tracked files the agent deleted (one path
per line). Empty/absent when nothing was deleted.
On run() (after the trial's docker env is up and /workspace has the base
state from the Dockerfile's COPY workspace/), the adapter:
1. Uploads agent-output/ into /workspace — overlays the agent's surviving
edits on top of the base workspace.
2. Applies _HARBOR_DELETIONS.txt by `rm -f`'ing each listed path under
/workspace. (The marker file itself was uploaded in step 1; it gets
removed too so the verifier's re-capture doesn't pick it up as
untracked content.)
3. Uploads trajectory.json to /logs/agent/trajectory.json so the grader
sees the same transcript it would have on the original run.
Verifier then runs as it would for any real trial — exact same code path,
exact same artifacts, just with the agent phase replaced by a deterministic
file-overlay. See scripts/harbor-regrade for the host-side wrapper.
Older reference runs captured before the deletion-capture line shipped
(verifier-deletion-capture PR #219) won't have _HARBOR_DELETIONS.txt; the
deletion-replay step is a no-op in that case. Deletions made in those runs
remain lost.
"""
from pathlib import Path, PurePosixPath
from harbor.agents.base import BaseAgent
from harbor.environments.base import BaseEnvironment
from harbor.models.agent.context import AgentContext
from harbor.models.trial.paths import EnvironmentPaths
# Sibling module (shipped alongside in the worker toolkit; on PYTHONPATH via
# harbor-regrade). Kept harbor-free so its logic stays unit-testable.
from reference_run_capture import captured_mutating_tools
_WORKSPACE = PurePosixPath("/workspace")
_DELETIONS_MARKER = "_HARBOR_DELETIONS.txt"
class ReplayAgent(BaseAgent):
"""Replays a captured reference run so the verifier can be re-graded
without invoking the model again."""
SUPPORTS_WINDOWS: bool = False # paths below assume POSIX /workspace.
def __init__(
self,
logs_dir,
reference_run_dir: str,
source_agent_import_path: str | None = None,
source_model_name: str | None = None,
**kwargs,
):
# `source_*` are provenance, not behaviour: harbor-regrade reads them off the
# source run and passes them so THIS replay's result.json records which
# harness and model produced the trajectory being graded. A replay reports
# `replay_agent:ReplayAgent` with model_name null, and the source run is
# usually deleted (a regrade is copied back over what it regraded), so a bare
# reference_run_dir pointer does not survive as provenance.
#
# They must be accepted here rather than left in **kwargs: harbor records the
# trial config's agent kwargs regardless of what the agent does with them, but
# BaseAgent would reject the unknown keys and take every regrade down with it.
self._source_agent_import_path = source_agent_import_path
self._source_model_name = source_model_name
super().__init__(logs_dir=logs_dir, **kwargs)
ref = Path(reference_run_dir).expanduser().resolve()
if not ref.is_dir():
raise FileNotFoundError(f"reference_run_dir does not exist: {ref}")
self._reference_run_dir = ref
self._agent_output_dir = ref / "agent-output"
self._trajectory_path = ref / "agent" / "trajectory.json"
@staticmethod
def name() -> str:
return "replay"
def version(self) -> str:
return "1.0.0"
async def setup(self, environment: BaseEnvironment) -> None:
# No installation needed; the verifier brings everything it requires.
return
async def run(
self,
instruction: str,
environment: BaseEnvironment,
context: AgentContext,
) -> None:
# Advisory tasks — the agent only reads and answers in chat — make NO
# workspace edits, so a faithful capture of one has an empty (or, in
# older pipelines, absent) agent-output/. That is not a data gap: the
# deliverable is the agent's final message, captured in
# agent/trajectory.json, which the grader reads via
# /tmp/outputs/task_transcript.txt. So overlay captured edits when
# present; otherwise grade the base workspace + transcript, exactly
# what the original advisory grading saw.
if not self._agent_output_dir.is_dir():
mutating = captured_mutating_tools(self._reference_run_dir)
if mutating:
# The agent edited files but they weren't captured — grading the
# base workspace would silently score the wrong state. Refuse.
raise FileNotFoundError(
f"reference_run_dir {self._reference_run_dir} has no "
f"agent-output/ but its captured transcript shows "
f"file-mutating tool calls {sorted(mutating)}. The agent's "
f"workspace edits were lost at capture time, so this run "
f"cannot be faithfully re-graded — re-capture it."
)
self.logger.warning(
"reference_run %s has no agent-output/ and made no "
"file-mutating tool calls — treating it as an advisory run and "
"grading the base workspace + captured transcript.",
self._reference_run_dir,
)
# Skip the overlay/deletion steps; fall through to trajectory upload.
await self._upload_trajectory(environment)
return
# 1. Overlay captured agent edits onto the base /workspace.
await environment.upload_dir(
source_dir=str(self._agent_output_dir),
target_dir=str(_WORKSPACE),
)
# 2. Apply captured deletions, if present. Read the marker from the
# host so we don't have to shell into the container to parse it,
# then issue per-path rm's plus a final cleanup of the marker
# itself (which was uploaded in step 1).
host_marker = self._agent_output_dir / _DELETIONS_MARKER
if host_marker.exists():
deletion_paths = [
line.strip()
for line in host_marker.read_text().splitlines()
if line.strip()
]
for raw in deletion_paths:
self._validate_relative_path(raw)
await environment.exec(
command=f'rm -f -- "/workspace/{raw}"',
user="root",
)
await environment.exec(
command=f'rm -f -- "/workspace/{_DELETIONS_MARKER}"',
user="root",
)
# 3. Materialize the captured trajectory at the path the grader's
# test.sh symlinks to /tmp/outputs/task_transcript.txt.
await self._upload_trajectory(environment)
async def _upload_trajectory(self, environment: BaseEnvironment) -> None:
"""Upload agent/trajectory.json to the path the grader's test.sh
symlinks to /tmp/outputs/task_transcript.txt. Shared by the normal
(overlay) path and the advisory (no agent-output) path."""
if self._trajectory_path.exists():
env_paths = EnvironmentPaths.for_os(environment.os)
await environment.upload_file(
source_path=str(self._trajectory_path),
target_path=str(env_paths.agent_dir / "trajectory.json"),
)
else:
# The grader's test.sh reads /tmp/outputs/task_transcript.txt,
# which symlinks to trajectory.json. Without the file the symlink
# dangles and the grader sees an empty transcript — so the regrade
# will look like the agent did nothing. Yell via harbor's own
# logger (self.logger is a child of harbor.utils.logger) so the
# warning lands in trial.log, not a stray "replay-agent" logger
# nothing's wired to.
self.logger.warning(
"reference_run %s has no agent/trajectory.json — the grader "
"will see an empty transcript. Investigate whether the source "
"run was produced by an older harbor that didn't write the "
"ATIF file (or by snapshot_agent before the multi-JSONL fix).",
self._reference_run_dir,
)
@staticmethod
def _validate_relative_path(raw: str) -> None:
"""Guard against absolute paths and `..` traversal in the deletions
manifest. The marker should only list paths *under* the workspace
root; anything else is a captured-data integrity problem worth
failing loudly on."""
if not raw or raw.startswith("/"):
raise ValueError(f"refusing absolute path in {_DELETIONS_MARKER}: {raw!r}")
if ".." in PurePosixPath(raw).parts:
raise ValueError(f"refusing `..` traversal in {_DELETIONS_MARKER}: {raw!r}")

View File

@@ -0,0 +1,459 @@
#!/usr/bin/env python3
"""resolve_harness.py — turn a harness id + task dir into the flags a trial needs.
``scripts/harbor-run`` is bash and cannot parse the TOML registry, so it shells in
here and evals the result::
RESOLVED="$(python3 scripts/resolve_harness.py --task-dir "$TASK_DIR")" || exit 1
eval "$RESOLVED"
Python rather than TS on purpose: this ships in the worker toolkit, whose
package.json has no ``zod``/``smol-toml``, and ``tomllib`` is stdlib — so the
toolkit gains a harness-aware harbor-run with zero new dependencies. There is no TS
loader: TS callers (submit-task) shell in here, so both the schema and the selection
policy exist exactly once and there is nothing to drift.
Output is POSIX ``KEY='value'`` assignments (single-quoted, embedded quotes
escaped) on stdout; everything human-facing goes to stderr, so the eval only ever
sees assignments. A non-zero exit means "do not launch" — the point is to fail in a
second rather than burn agent minutes on a trial that cannot produce a usable grade.
Refuses to resolve when:
- the harness id is unknown or disabled
- the harness writes no ATIF trajectory (the grader would have no transcript)
- the task ships a session to resume but the harness cannot resume one. This is
the important one: it is the only failure here that would otherwise look like
SUCCESS, with the agent answering a prompt whose conversation it never saw.
- the harness's credential env var is unset
``--check-model`` additionally asks the proxy whether the model is granted. Opt-in
on purpose: it is a network call, and one in every run's critical path trades a fast
local failure for a new way to hang. The credential check, which is free, always runs.
"""
from __future__ import annotations
import argparse
import json
import os
import re
import shlex
import sys
import tomllib
import urllib.error
import urllib.request
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
from harness_registry import ( # noqa: E402
Harness,
HarnessRegistryError,
load_harness_registry,
)
# Harness used when nothing selects one. Keeps every existing caller on today's
# behaviour, so adding harness selection changes no current run.
DEFAULT_HARNESS = "claude-code"
MODELS_TIMEOUT_SEC = 20
def warn(message: str) -> None:
print(f"resolve-harness: {message}", file=sys.stderr)
def fail(message: str) -> "None":
print(f"resolve-harness: ERROR: {message}", file=sys.stderr)
raise SystemExit(1)
def is_multi_turn(task_dir: str | None) -> bool:
"""A task is multi-turn when it ships a NON-EMPTY session to resume. Empty is
the documented one-shot-snapshot fallback and must run cold, so size is the
test, not existence."""
if not task_dir:
return False
session = Path(task_dir) / "environment" / "session.jsonl"
return session.is_file() and session.stat().st_size > 0
def wants_browser(task_dir: str | None) -> bool:
"""True when task.toml opts into a browser (`[metadata] browser = true`).
Read straight from the file rather than via tomllib: this must agree with
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
text match. If the two ever disagree the agent is told about a browser the image
lacks, which is the one failure the disclosure is designed to make impossible.
Accepts the quoted form for the same reason build-workspace.sh does."""
if not task_dir:
return False
toml_path = Path(task_dir) / "task.toml"
if not toml_path.is_file():
return False
try:
text = toml_path.read_text(encoding="utf-8")
except OSError:
return False
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
def harness_from_task_toml(task_dir: str | None) -> str | None:
"""The task's own `[agent] harness` — the authoritative record of which harness
this task was authored against.
This is where the worker's choice lands: the snapshot flow stamps it from the CLI
that produced the snapshot, and a manual author writes it themselves. Either way
it is set at task-creation time, BEFORE any trial, so nothing here depends on a
trial's output.
Parsed with tomllib rather than a grep: a regex would happily match a commented
line or the wrong table, and picking the wrong harness is a silent
wrong-agent-runs bug.
Returns None when the field is simply absent — the normal case for every task
finalized before harness selection existed — so the caller falls through to the
toolkit default.
But an UNPARSEABLE task.toml refuses outright rather than falling back. Those are
different situations and treating them alike is how the wrong harness runs
quietly: the most likely way to break this file is adding a second `[agent]`
table instead of a `harness` line inside the existing one (tasks already carry
`[agent] timeout_sec`), and TOML rejects a duplicate table. Falling back there
would run claude against a task its author wrote for codex and grade it as if
nothing were wrong.
"""
if not task_dir:
return None
path = Path(task_dir) / "task.toml"
if not path.is_file():
return None
try:
with open(path, "rb") as handle:
doc = tomllib.load(handle)
except (OSError, tomllib.TOMLDecodeError) as exc:
fail(
f"{path} could not be parsed ({exc}). Refusing to guess a harness — fix "
f'the file. If you were adding a harness, put `harness = "..."` inside '
f"the EXISTING [agent] table rather than starting a second one."
)
harness = (doc.get("agent") or {}).get("harness")
return harness if isinstance(harness, str) and harness else None
def normalize_model(harness: Harness, model: str) -> str:
"""Model id on the wire, per the harness's declared shape."""
if harness.model_id_shape == "provider:model":
return model.replace("/", ":")
return model
def granted_models(harness: Harness) -> list[str] | None:
"""Model ids the key is granted, or None when the check couldn't run."""
base_url = os.environ.get(harness.base_url_env or "")
key = os.environ.get(harness.key_env or "")
if not base_url or not key:
warn("--check-model skipped: base URL or key env is unset")
return None
request = urllib.request.Request(
f"{base_url.rstrip('/')}/models", headers={"Authorization": f"Bearer {key}"}
)
try:
with urllib.request.urlopen(request, timeout=MODELS_TIMEOUT_SEC) as response:
body = json.loads(response.read().decode("utf-8"))
except (urllib.error.URLError, TimeoutError, ValueError, OSError) as exc:
warn(f"--check-model skipped: /models unreachable ({exc})")
return None
return [m["id"] for m in body.get("data", []) if isinstance(m.get("id"), str)]
def assert_model_granted(harness: Harness, model: str) -> None:
granted = granted_models(harness)
if granted is None:
return
# The proxy LISTS ids provider-prefixed ("openai/gpt-5.6-sol") but 400s on that
# form — requests take the bare id. Accept either spelling.
bare = {g.split("/")[-1] for g in granted}
if model not in granted and model not in bare:
shown = ", ".join(granted[:12]) + (", …" if len(granted) > 12 else "")
fail(
f'Model "{model}" is not granted for this key. Granted ({len(granted)}): {shown}'
)
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
parser.add_argument(
"--harness",
help=f"harness id (default: the task's [agent] harness, else {DEFAULT_HARNESS})",
)
parser.add_argument(
"--task-dir",
help="task directory; decides multi-turn from environment/session.jsonl",
)
parser.add_argument("--model", help="override the harness's default model")
parser.add_argument(
"--fast",
action="store_true",
help="run the trial agent in the harness's fast serving mode (higher token "
"rate, faster output). Refuses on a harness that has none.",
)
parser.add_argument(
"--check-model",
action="store_true",
help="also ask the proxy whether the model is granted (network call)",
)
parser.add_argument(
"--authoring-installs",
action="store_true",
help="print '<id>\\t<cli>\\t<install>' for each harness a worker can author "
"with, and exit. Consumed by scripts/setup-harnesses.sh so the worker "
"containers install from the registry rather than from hardcoded lists that "
"drift.",
)
parser.add_argument(
"--container-configs",
action="store_true",
help="print '<id>\\t<config_path>\\t<base64 container_config>' for each "
"authoring harness that declares one, and exit. Base64 because the config is "
"multi-line TOML and these query modes are line-oriented.",
)
parser.add_argument(
"--surface",
choices=("authoring", "explore"),
default="authoring",
help="which worker container --container-configs is for; explore additionally "
"gets the capture hooks, whose commands only ship there.",
)
parser.add_argument(
"--defaults",
action="store_true",
help="print '<id>\\t<default_model>\\t<effort_default>' for every harness, and "
"exit. For recording what a task was authored against; nothing reads it back.",
)
parser.add_argument(
"--explore-launchers",
action="store_true",
help="print '<id>\\t<cli>\\t<launch command>' for each authoring harness, and "
"exit. The launch command has the registry's model, effort and agent_config "
"already substituted, so Explore and a trial cannot disagree about them. "
"Consumed by setup-harnesses.sh to write one launcher per harness.",
)
parser.add_argument(
"--skills-dirs",
action="store_true",
help="print '<id>\t<skills_dir>' for each authoring harness that discovers "
"skills from a directory, and exit. Lets setup-harnesses.sh install the "
"snapshot skill for harnesses that have no plugin system.",
)
parser.add_argument(
"--auth-files",
action="store_true",
help="print '<id>\t<auth_path>\t<auth_key_env>' for each authoring harness that "
"authenticates from a file rather than the environment, and exit.",
)
parser.add_argument(
"--authoring-credentials",
action="store_true",
help="print '<id>\\t<key_env>\\t<base_url_env>\\t<proxy_path>' for each "
"authoring harness, and exit. Lets the containers point every harness at the "
"same proxy key on its own provider path.",
)
parser.add_argument(
"--declared-harness",
action="store_true",
help="print ONLY the harness --task-dir's task.toml declares (empty if it "
"declares none) and exit. Unlike the default mode this applies no fallback, so "
"a caller can tell 'declared' from 'defaulted'. Exists so consumers without a "
"TOML parser never hand-roll one: a regex would match a commented line or the "
"wrong table, and the duplicate-[agent] shape is exactly the likely mistake.",
)
parser.add_argument(
"--resolve-identity",
action="append",
default=None,
metavar="AGENT",
help="resolve agent identities (a result.json config.agent import_path or name) "
"to harness ids and exit; repeatable. Prints one TAB-separated "
"'<identity>\\t<harness-id>' line each, with an empty id when nothing claims it. "
"Lets callers that cannot import the registry (the worker toolkit has no "
"zod/smol-toml) still resolve through the one source of truth.",
)
parser.add_argument(
"--list",
action="store_true",
help="print the selectable harnesses and exit (what task.toml's [agent] harness accepts)",
)
parser.add_argument("--registry", default=None, help="registry path (tests)")
args = parser.parse_args(argv)
try:
registry = (
load_harness_registry(args.registry)
if args.registry
else load_harness_registry()
)
except HarnessRegistryError as exc:
fail(str(exc))
# --- read-only query modes: answer and exit, never emit assignments -------
if args.authoring_installs:
for harness in registry.authoring():
print(f"{harness.id}\t{harness.cli or ''}\t{harness.install or ''}")
return 0
if args.container_configs:
import base64
for harness in registry.authoring():
config = harness.container_config_text(surface=args.surface)
if not (harness.config_path and config):
continue
blob = base64.b64encode(config.encode()).decode()
print(f"{harness.id}\t{harness.config_path}\t{blob}")
return 0
if args.defaults:
for harness in registry.all():
print(
f"{harness.id}\t{harness.default_model or ''}\t"
f"{harness.effort_default or ''}"
)
return 0
if args.explore_launchers:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.cli or ''}\t"
f"{harness.explore_launch_command() or ''}"
)
return 0
if args.skills_dirs:
for harness in registry.authoring():
if harness.skills_dir:
print(f"{harness.id}\t{harness.skills_dir}")
return 0
if args.auth_files:
for harness in registry.authoring():
if harness.auth_path and harness.auth_key_env:
print(f"{harness.id}\t{harness.auth_path}\t{harness.auth_key_env}")
return 0
if args.authoring_credentials:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.key_env or ''}\t"
f"{harness.base_url_env or ''}\t{harness.proxy_path or ''}"
)
return 0
if args.declared_harness:
print(harness_from_task_toml(args.task_dir) or "")
return 0
if args.resolve_identity:
for identity in args.resolve_identity:
harness = registry.by_import_path(identity)
print(f"{identity}\t{harness.id if harness else ''}")
return 0
if args.list:
# Printed on stdout because it is the requested output here, not the
# eval-able assignments — this mode is for a human, and never shelled into.
for harness in registry.enabled():
turns = (
"multi-turn + single-turn"
if harness.seed_native
else "single-turn only"
)
model = harness.default_model or "(pass --model)"
print(f"{harness.id:<16} {harness.label:<20} {turns:<24} {model}")
return 0
# --- selection ------------------------------------------------------------
# Precedence: an explicit --harness (a benchmark, or a deliberate override) beats the
# task's own record, which is what its author chose. Everything else — every task
# finalized before harness selection existed — is the default.
requested = args.harness or harness_from_task_toml(args.task_dir) or DEFAULT_HARNESS
try:
harness = registry.require(requested)
except HarnessRegistryError as exc:
fail(str(exc))
if not harness.enabled:
fail(
f'Harness "{harness.id}" is disabled in the registry (never verified here). '
f"Enable it in scripts/harness-registry.toml once a trial has been run with it."
)
if not harness.writes_atif:
fail(
f'Harness "{harness.id}" writes no ATIF trajectory, so the grader would have '
f"no transcript and its rewards would be meaningless."
)
multi_turn = is_multi_turn(args.task_dir)
if multi_turn and not harness.seed_native:
fail(
f'Task ships a session to resume, but harness "{harness.id}" cannot resume '
f"one. Running anyway would look like a success while the agent answered a "
f"prompt whose conversation it never saw."
)
if harness.key_env and not os.environ.get(harness.key_env):
fail(f'{harness.key_env} is unset — required by harness "{harness.id}".')
if args.fast and not harness.fast_kwarg:
fail(
f'Harness "{harness.id}" has no fast serving mode (no fast_kwarg in the '
f"registry). Drop --fast or pick a harness that declares one."
)
model = args.model or harness.default_model
if not model:
fail(
f'Harness "{harness.id}" has no default_model in the registry; pass --model '
f"explicitly."
)
if args.check_model:
assert_model_granted(harness, model)
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
# anything it would only echo back at the worker is said below instead.
browser = wants_browser(args.task_dir)
assignments = {
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
"MODEL": normalize_model(harness, model),
"EFFORT_KWARG": harness.effort_kwarg,
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
"FAST_KWARG": harness.fast_kwarg if args.fast else "",
}
warn(
f"{harness.label} · model={assignments['MODEL']} · "
f"{'multi-turn' if multi_turn else 'single-turn'} · "
f"{'browser · ' if browser else ''}"
f"{'fast · ' if args.fast else ''}"
f"agent={assignments['AGENT_IMPORT_PATH']}"
)
if browser and not harness.agent_import_path_browser:
# Not a failure: the image still gets Playwright and the agent is still told about
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
# letting someone infer from a log line that the opt-in was ignored entirely.
warn(
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
f"The browser and its disclosure are unaffected."
)
if harness.flaky_hangs:
warn(
f"{harness.label} is known to hang with no client-side timeout on a small "
f"fraction of trials. A silent, output-less trial is that, not a task defect."
)
for key, value in assignments.items():
print(f"{key}={shlex.quote(value)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,260 @@
/**
* Strip machine-identifying filesystem paths, and optional keywords, from a session
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
*/
export const DEFAULT_PLACEHOLDER = '~/repo';
export const HOME_DIR_PLACEHOLDER = '~';
export const REDACTION_PLACEHOLDER = '[redacted]';
export interface SanitizeOptions {
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
placeholder?: string;
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
forbiddenMarkers?: readonly RegExp[];
/**
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
* there, so callers that know the root pass it here.
*/
cwdPrefix?: string;
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
cwdPrefixes?: readonly string[];
/**
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
*/
scrubEmbeddedHomePaths?: boolean;
}
export interface SanitizeResult {
sanitized: string;
prefixStripped: string | null;
encodedPrefixStripped: string | null;
homeDirStripped: string | null;
encodedHomeDirStripped: string | null;
embeddedPrefixStripped: string | null;
embeddedHomeDirStripped: string | null;
/** Replacement count per marker, keyed by the regex's source string. */
markersScrubbed: Record<string, number>;
}
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
* Returns `''` when only the root `/` is common. */
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
const arr = Array.from(paths);
if (arr.length === 0) return '';
const splits = arr.map((p) => p.split('/'));
const minLen = Math.min(...splits.map((s) => s.length));
let lastShared = 0;
for (let i = 0; i < minLen; i++) {
const c = splits[0][i];
if (splits.some((s) => s[i] !== c)) break;
lastShared = i + 1;
}
// Only the leading empty piece matched → just the root, not useful.
if (lastShared <= 1) return '';
return splits[0].slice(0, lastShared).join('/');
}
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
* better to skip the home pass than strip what may be repo content. */
export function extractHomeDir(cwdPrefix: string): string | null {
if (!cwdPrefix.startsWith('/')) return null;
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
// drive letter and leave the account name in. A volume or drive root carries no
// identity by itself, so those take the directory under it.
const patterns: RegExp[] = [
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
/^\/mnt\/[^/]+\/Users\/[^/]+/,
/^\/Users\/[^/]+/,
/^\/home\/[^/]+/,
/^\/Volumes\/[^/]+\/[^/]+/,
/^\/mnt\/[^/]+\/[^/]+/,
/^\/var\/root(?=\/|$)/,
/^\/root(?=\/|$)/,
];
for (const re of patterns) {
const m = cwdPrefix.match(re);
if (m) return m[0];
}
return null;
}
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
export function collectCwds(raw: string): Set<string> {
const out = new Set<string>();
const add = (v: unknown) => {
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
};
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const rec = parsed as { cwd?: unknown; payload?: unknown };
add(rec.cwd);
if (typeof rec.payload === 'object' && rec.payload !== null) {
add((rec.payload as { cwd?: unknown }).cwd);
}
}
return out;
}
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
// macOS/Windows display names can contain spaces, but only consume them while
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
const EMBEDDED_HOME_RE = new RegExp(
'(?:' +
String.raw`\/home\/${COMP}` +
'|' +
String.raw`\/Users\/${USER_WITH_SPACES}` +
'|' +
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
'|' +
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
String.raw`\/var\/root(?![^/])` +
'|' +
String.raw`\/root(?![^/])` +
')' +
String.raw`(?:\/${COMP})*`,
'g'
);
export function collectEmbeddedHomePaths(raw: string): Set<string> {
const out = new Set<string>();
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
return out;
}
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
return haystack.split(needle).join(replacement);
}
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
* is one component but `…/repo.` ending a sentence is not. */
function continuesComponent(text: string, at: number): boolean {
const ch = text[at];
if (ch === undefined) return false;
if (/[A-Za-z0-9_-]/.test(ch)) return true;
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
}
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
let out = '';
let from = 0;
for (;;) {
const i = haystack.indexOf(needle, from);
if (i === -1) return out + haystack.slice(from);
const end = i + needle.length;
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
from = end;
}
}
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
function stripBothForms(haystack: string, needle: string, replacement: string): string {
const out = literalReplaceAll(haystack, needle, replacement);
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
}
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
const markers = opts.forbiddenMarkers ?? [];
const cwds = collectCwds(raw);
let working = raw;
let prefixStripped: string | null = null;
let encodedPrefixStripped: string | null = null;
let homeDirStripped: string | null = null;
let encodedHomeDirStripped: string | null = null;
let embeddedPrefixStripped: string | null = null;
let embeddedHomeDirStripped: string | null = null;
const requested = opts.cwdPrefixes?.length
? [...opts.cwdPrefixes]
: opts.cwdPrefix
? [opts.cwdPrefix]
: cwds.size > 0
? [findLongestCommonPathPrefix(cwds)]
: [];
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
// sibling root's own prefix, leaving it unmatched when its turn came.
for (const prefix of prefixes) {
const encodedPrefix = prefix.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, prefix, placeholder);
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
prefixStripped ??= prefix;
encodedPrefixStripped ??= encodedPrefix;
}
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
const homeDirs = new Set(
prefixes
.map((p) => extractHomeDir(p))
.filter((h): h is string => h !== null && !prefixes.includes(h))
);
for (const homeDir of homeDirs) {
const encodedHomeDir = homeDir.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
homeDirStripped ??= homeDir;
encodedHomeDirStripped ??= encodedHomeDir;
}
if (opts.scrubEmbeddedHomePaths) {
const embedded = collectEmbeddedHomePaths(working);
if (embedded.size > 0) {
// Take each path's own shortest `/repo`-terminated prefix rather than a
// common prefix, which mis-collapses when paths diverge above the root.
const repoRoots = new Set<string>();
const homeDirs = new Set<string>();
for (const p of embedded) {
const h = extractHomeDir(p);
if (h) homeDirs.add(h);
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
if (m) repoRoots.add(m[1]);
}
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
embeddedPrefixStripped = sortedRoots[0] ?? null;
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
}
}
const markersScrubbed: Record<string, number> = {};
for (const re of markers) {
let count = 0;
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
const global = new RegExp(re.source, flags);
working = working.replace(global, () => {
count++;
return REDACTION_PLACEHOLDER;
});
if (count > 0) markersScrubbed[re.source] = count;
}
return {
sanitized: working,
prefixStripped,
encodedPrefixStripped,
homeDirStripped,
encodedHomeDirStripped,
embeddedPrefixStripped,
embeddedHomeDirStripped,
markersScrubbed,
};
}

View File

@@ -0,0 +1,34 @@
/**
* Shared helper for finding a Claude Code session id inside a JSONL or
* stream-json `.txt` file. Both formats embed the id under one of two keys:
*
* - `session_id` — stream-json output (`claude-code.txt` from `--print
* --output-format=stream-json`).
* - `sessionId` — Claude Code's internal session log (the resumable JSONL
* under `~/.claude/projects/<key>/<id>.jsonl`).
*
* Real files only use one key, but if a future format ever emits both we
* shouldn't have two callers picking different winners — so this helper is
* the single source of truth.
*/
import { readFileSync } from 'fs';
export function readSessionId(filePath: string): string | null {
const lines = readFileSync(filePath, 'utf8').split('\n');
for (const line of lines) {
if (!line) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const obj = parsed as Record<string, unknown>;
const camel = typeof obj.sessionId === 'string' ? obj.sessionId : null;
const snake = typeof obj.session_id === 'string' ? obj.session_id : null;
const id = camel ?? snake;
if (id) return id;
}
return null;
}

View File

@@ -0,0 +1,340 @@
#!/bin/bash
# Install the harnesses a worker can author with, from scripts/harness-registry.toml.
#
# Source it, then call unpiped — it exports credentials, which a subshell would lose:
#
# . /workspace/scripts/setup-harnesses.sh
# harness_setup_all
#
# Registry reading and credential derivation live in lib/harness-credentials.sh, sourced
# below, because `harbor-run` needs those and nothing else here.
#
# No -e here — but this file is SOURCED, and shell options belong to the caller's shell:
# both post-creates run with -e, so that is what is in force. An unguarded failure below
# therefore aborts container creation, which is why every failure site is individually
# guarded (`|| true`, `if !`) rather than relying on this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
if [ ! -f "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh" ]; then
echo "harness-setup: FATAL — $_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh is" >&2
echo "harness-setup: missing, so nothing here can read the registry. Every step below" >&2
echo "harness-setup: would report a missing interpreter instead of this." >&2
return 1 2>/dev/null || exit 1
fi
# shellcheck disable=SC1091
. "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh"
# Every setup step reads the registry through _harness_query, and each call suppresses
# stderr so one bad row can't abort the container. That means a BROKEN interpreter turns
# the whole of setup into a silent no-op: no credentials, no CLIs, no config, no
# launchers, and no error anywhere. Check it once, loudly, before any of that.
harness_preflight() {
local err py found=yes
py=$(_raccoon_python) || { py=python3; found=no; }
if ! err=$("$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" --list 2>&1 >/dev/null); then
echo "harness-setup: FATAL — cannot read the harness registry, so no agent CLI" >&2
echo "harness-setup: would be installed. Nothing below will run." >&2
echo "harness-setup: interpreter: $(command -v "$py" || echo MISSING) ($("$py" -V 2>&1))" >&2
if [ "$found" = no ]; then
echo "harness-setup: no python3.11+ with tomllib found; set RACCOON_PYTHON to override" >&2
fi
echo "harness-setup: registry: $_HARNESS_REGISTRY_DIR/harness-registry.toml" >&2
printf 'harness-setup: %s\n' "$err" >&2
return 1
fi
}
# claude installs into $HOME/.local/bin, which is not on PATH during post-create.
case ":$PATH:" in
*":$HOME/.local/bin:"*) ;;
*) export PATH="$HOME/.local/bin:$PATH" ;;
esac
# --- installs ----------------------------------------------------------------
harness_install_clis() {
local id cli install
while IFS=$'\t' read -r id cli install; do
[ -n "$install" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo "harness-setup: $cli already installed — skipping" >&2
continue
fi
echo "harness-setup: installing $id ($cli)" >&2
# Reported as unavailable below rather than fatal.
if ! bash -c "$install" >&2; then
echo "harness-setup: WARNING $id failed to install — $cli will be unavailable" >&2
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
}
# Report which CLIs are usable. Non-zero when NONE are: one harness missing is survivable
# (a worker uses the other), but zero means the container cannot author anything at all,
# and that must stop setup rather than read as a couple of warnings.
harness_report() {
local id cli install ready=0 missing=0
while IFS=$'\t' read -r id cli install; do
[ -n "$cli" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo " $cli — ready" >&2
ready=$((ready + 1))
else
echo " $cli — NOT AVAILABLE (install failed; see above)" >&2
missing=$((missing + 1))
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
# A CLI on PATH with no key is worse than a missing one: it starts, then fails at the
# first request with the harness's own auth error, which says nothing about setup.
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
if [ -z "${!key_env:-}" ]; then
echo " $id — installed but NO CREDENTIALS: $key_env is unset." >&2
echo " Derived from ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY; set both in .env." >&2
fi
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
if [ "$ready" -eq 0 ]; then
echo "harness-setup: FATAL — no agent CLI installed ($missing attempted)." >&2
echo "harness-setup: This container cannot author a task. Check the install" >&2
echo "harness-setup: output above: the CLIs download over the network, so a" >&2
echo "harness-setup: proxy, DNS or upstream change breaks every one at once." >&2
return 1
fi
[ "$missing" -gt 0 ] && echo "harness-setup: $missing harness(es) unavailable; $ready usable" >&2
return 0
}
# --- Explore launchers -------------------------------------------------------
# One `raccoon-explore-<cli>` per harness, aliased to its `cli`.
harness_install_launchers() {
local bin="$HOME/.local/bin"
mkdir -p "$bin"
# Read at launcher run time so the note stays a file, not a baked-in copy.
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
local browser_note_src="${note_src%.md}_browser.md"
local read_note_src="${note_src%.md}_read.md"
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
# Which harnesses keep their key in a file rather than reading $ENV per request. Those
# launchers refresh it first: the file dates from container create, so a key rotated in
# .env since then would otherwise reach the harness only after a rebuild.
local file_auth_ids="" aid apath akey
while IFS=$'\t' read -r aid apath akey; do
[ -n "$apath" ] || continue
file_auth_ids="${file_auth_ids:+$file_auth_ids }$aid"
done < <(_harness_query --auth-files 2>/dev/null || true)
local id cli launch switchable refresh_line
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
# `|| true` twice over (here and inside the script): the launcher runs under
# `set -e`, and a failed refresh must not cost the worker their agent.
if [[ " $file_auth_ids " == *" $id "* ]]; then
refresh_line="\"$_HARNESS_REGISTRY_DIR/refresh-harness-auth\" || true"
else
refresh_line=""
fi
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
# so a browser task needs nothing added and the flag has nothing to switch.
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
# which every launch line references, and every harness would look switchable.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
switchable=1
else
switchable=0
fi
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
set -euo pipefail
if [ -f "$note_src" ]; then
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
else
RACCOON_TOOLSET_NOTE=""
fi
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
# here. Per invocation, not per container — authoring a browser task shouldn't need a
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
# still mirrors an ordinary trial.
#
# The correction must be appended AFTER the base note, which says there is no Read tool.
RACCOON_TOOLS="Bash"
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
RACCOON_TOOLS="Bash,Read"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\$(cat "$read_note_src")"
fi
export RACCOON_TOOLS
# Only mention the browser on an image that actually has one — most don't. Probed at
# launch, not baked in, so the same launcher is correct in whichever container it runs.
#
# Exported two ways because the harnesses take extra instructions differently: claude
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
# (it has no str_replace_editor), so the browser part is exported on its own too.
RACCOON_BROWSER_NOTE=""
RACCOON_BROWSER_FLAGS=()
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\${RACCOON_BROWSER_NOTE}"
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
fi
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
export RACCOON_HARNESS="$id"
# These launchers exist only in explore, and a refresh that has to CREATE a config
# needs the surface to know the capture hooks belong in it.
export RACCOON_SURFACE=explore
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
# — it has no CLAUDE_PLUGIN_* to fall back to. Exporting it ALSO overrode the dir for
# claude, whose slash command pins --plugin-data to the plugin dir, so the hook wrote one
# place and capture read another and the recorded session was silently ignored.
$refresh_line
$launch
LAUNCHER
chmod +x "$bin/raccoon-explore-$cli"
echo "harness-setup: launcher raccoon-explore-$cli" >&2
done < <(_harness_query --explore-launchers 2>/dev/null || true)
}
# Alias lines for ~/.bashrc.
harness_alias_lines() {
local id cli launch switchable
local browser_clis=""
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
echo "alias $cli=\"raccoon-explore-$cli\""
# Same derivation as the launcher: only a harness whose launch line takes
# $RACCOON_TOOLS has a toolset the flag can change.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
browser_clis="${browser_clis:+$browser_clis }$cli"
fi
done < <(_harness_query --explore-launchers 2>/dev/null || true)
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
# alternate screen buffer, so anything printed just before exec is hidden for the whole
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
#
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
# and in one without.
[ -n "$browser_clis" ] || return 0
local first="${browser_clis%% *}"
cat <<HINT
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
fi
HINT
}
# Write each harness's config file from the registry, replacing whatever was there.
#
# The file is OWNED, not merged: TOML has no way to return to the document root after a
# table header, so appending or prepending around foreign content silently reparents
# root-level keys into whichever table happens to precede them. Owning it also means a
# registry change actually reaches a container that was already set up.
harness_write_configs() {
local id config_path blob target tmp
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
# Guarded: a bare failing assignment exits the caller's `set -e` post-create with
# no explanation. A path this cannot expand is one harness's problem, not the
# container's.
target=$(eval "printf '%s' \"$config_path\"") || {
echo "harness-setup: WARNING $id config_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")"
tmp="$target.raccoon-tmp"
# Expansion is strict: an unset var would otherwise be written through as the
# literal ${VAR}, which surfaces much later as an unparseable value.
if ! {
echo "# Generated from harness-registry.toml — edits here are overwritten."
printf '%s' "$blob" | base64 -d | python3 -c '
import os, re, sys
text = sys.stdin.read()
missing = sorted(
{m.group(1) for m in re.finditer(r"\$\{(\w+)\}", text) if m.group(1) not in os.environ}
)
if missing:
sys.stderr.write("unset: " + ", ".join(missing) + "\n")
raise SystemExit(1)
sys.stdout.write(os.path.expandvars(text))
'
} > "$tmp"; then
rm -f "$tmp"
echo "harness-setup: WARNING $id config NOT written — a value it needs is unset." >&2
echo "harness-setup: run harness_setup_credentials first (harness_setup_all does)." >&2
continue
fi
mv "$tmp" "$target"
echo "harness-setup: $id config -> $target" >&2
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}
# Link every available skill into each harness's skills_dir, for harnesses that declare one.
# Both container layouts are covered: the explore container holds the snapshot skill under
# plugins/, the authoring container holds the authoring skills under .claude/skills. Whichever
# directories exist here are the ones this container has.
harness_install_skills() {
local sources="${RACCOON_SKILL_SOURCE_DIRS:-/workspace/plugins/create-snapshot/skills /workspace/.claude/skills}"
local id dir target src skill name installed
while IFS=$'\t' read -r id dir; do
[ -n "$dir" ] || continue
target=$(eval "printf '%s' \"$dir\"") || {
echo "harness-setup: WARNING $id skills_dir could not be expanded — skipping" >&2
continue
}
mkdir -p "$target"
installed=0
for src in $sources; do
[ -d "$src" ] || continue
for skill in "$src"/*/; do
[ -f "$skill/SKILL.md" ] || continue
name=$(basename "$skill")
ln -sfn "${skill%/}" "$target/$name"
installed=$((installed + 1))
done
done
echo "harness-setup: $id skills -> $target ($installed linked)" >&2
done < <(_harness_query --skills-dirs 2>/dev/null || true)
}
# The lines that explain a setup failure are printed as it happens, and the devcontainer
# CLI's own stack trace lands on top of them. Close with a banner so the worker has
# something to look for, and something to send us.
_harness_fatal_banner() {
echo "" >&2
echo " ============================================================" >&2
echo " HARNESS SETUP FAILED — this container has no agent CLI." >&2
echo "" >&2
echo " The harness-setup: lines above say why. Anything the" >&2
echo " devcontainer prints after this is a consequence, not the" >&2
echo " cause; send us the harness-setup: lines." >&2
echo " ============================================================" >&2
echo "" >&2
}
harness_setup_all() {
harness_preflight || { _harness_fatal_banner; return 1; }
harness_setup_credentials
harness_write_auth
harness_install_clis
harness_write_configs
harness_install_skills
# Launchers are NOT installed here. They are an Explore concern (that container aliases
# `claude`/`codex` to them), and it passes its own AGENT_CLI_DIR — installing them here
# too wrote every launcher twice, the first time with the wrong editor path, and left an
# unused one in the authoring container.
echo "harness-setup: authoring harnesses" >&2
harness_report || { _harness_fatal_banner; return 1; }
}

View File

@@ -0,0 +1,837 @@
/**
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
*
* Usage:
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
*/
import { execFileSync, execSync } from 'child_process';
import {
chmodSync,
copyFileSync,
existsSync,
mkdirSync,
readFileSync,
readdirSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
// --- CLI ---
const argv = yargs(hideBin(process.argv))
.option('snapshot', {
type: 'string',
describe: 'Path to the snapshot directory',
demandOption: true,
})
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'snapshot-to-task', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
// --- Read snapshot data ---
const snapshotDir = argv.snapshot;
if (!existsSync(snapshotDir)) {
log.fatal(
{ path: snapshotDir },
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
);
process.exit(1);
}
interface SnapshotMetadata {
slug: string;
session_uuid: string;
/** Absent on snapshots captured before harness selection existed. */
harness?: string;
original_cwd: string;
commit: string | null;
branch: string | null;
remote_url: string | null;
timestamp: string;
plugin_version: string;
}
interface Annotation {
what_trying: string;
what_hoping: string;
what_happened: string;
[key: string]: string;
}
const metadata = JSON.parse(
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
) as SnapshotMetadata;
const annotation = JSON.parse(
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
) as Annotation;
if (!metadata.slug) {
log.fatal(
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
);
process.exit(1);
}
const slug = metadata.slug;
// --- Locate harbor infrastructure ---
function findRepoRoot(): string | null {
let dir = process.cwd();
while (dir !== resolve(dir, '..')) {
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
dir = resolve(dir, '..');
}
return null;
}
const maybeRepoRoot = findRepoRoot();
if (!maybeRepoRoot) {
log.fatal(
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
);
process.exit(1);
}
const repoRoot: string = maybeRepoRoot;
const harborTasks = join(repoRoot, 'harbor-tasks');
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
const sharedDir = sharedCandidates.find((d) => existsSync(d));
const taskDir = join(harborTasks, slug);
if (existsSync(taskDir)) {
log.fatal(
{ path: taskDir },
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
);
process.exit(1);
}
if (!sharedDir) {
log.fatal(
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
);
process.exit(1);
}
// --- Detect repo name ---
interface ToolkitConfig {
repo: string;
defaultCommit: string;
/** The packed kit's release version (git describe at pack time). */
version?: string;
}
function readToolkitConfig(): ToolkitConfig | null {
const configPath = join(repoRoot, 'toolkit.json');
if (!existsSync(configPath)) return null;
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
}
function repoNameFromRemote(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
return match ? match[1] : null;
}
function findSubmoduleDir(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const reposDir = join(repoRoot, 'repos');
if (!existsSync(reposDir)) return null;
const normalize = (url: string) =>
url
.replace(/\.git$/, '')
.replace(/^git@github\.com:/, 'https://github.com/')
.toLowerCase();
for (const entry of readdirSync(reposDir)) {
const repoPath = join(reposDir, entry, 'repo');
if (!existsSync(repoPath)) continue;
try {
const remote = execSync('git remote get-url origin', {
cwd: repoPath,
encoding: 'utf8',
stdio: ['pipe', 'pipe', 'pipe'],
}).trim();
if (normalize(remote) === normalize(remoteUrl)) return entry;
} catch {
continue;
}
}
return null;
}
const toolkitConfig = readToolkitConfig();
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
// Derive which member this task targets from the snapshot's original_cwd basename,
// validated against the member list.
const polyglotMember = (() => {
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
const members = cfg.repos.map((r) => r.repo);
return base && members.includes(base) ? base : null;
})();
const repoName =
polyglotMember ??
toolkitConfig?.repo ??
findSubmoduleDir(metadata.remote_url) ??
repoNameFromRemote(metadata.remote_url);
if (!repoName) {
log.fatal(
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
);
process.exit(1);
}
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
const sessionUuid = metadata.session_uuid;
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
// --- Create task directory structure ---
mkdirSync(join(taskDir, 'environment'), { recursive: true });
mkdirSync(join(taskDir, 'tests'), { recursive: true });
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// --- Copy shared infrastructure ---
// The complete grader asset set test.sh depends on: the grader system prompt
// and the renderer (test.sh exits without the renderer). Sources missing from
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
},
// test.sh execs this to render the grade; without it the verifier writes no reward
// file and the trial errors out rather than scoring.
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
];
for (const { src, dest } of sharedFiles) {
const srcPath = join(sharedDir, src);
const destPath = join(taskDir, dest);
if (existsSync(srcPath)) {
copyFileSync(srcPath, destPath);
if (src === 'test.sh') chmodSync(destPath, 0o755);
log.debug({ src, dest }, 'Copied shared file');
} else {
log.warn({ src }, 'Shared file not found');
}
}
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
// their output to the grader as evidence for the CORRECTNESS score, so without
// them a code task's correctness is never signal-backed — the grader falls back
// to reading the diff alone. Same per-member-then-generic resolution as the
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
// member, a single-repo toolkit ships the lone test-commands.sh.
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
const genericTestCommands = join(sharedDir, 'test-commands.sh');
const testCommandsSrc = existsSync(perMemberTestCommands)
? perMemberTestCommands
: genericTestCommands;
if (existsSync(testCommandsSrc)) {
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
copyFileSync(testCommandsSrc, testCommandsDest);
chmodSync(testCommandsDest, 0o755);
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
} else {
// Not fatal: the grader still scores correctness by walking the changed code.
log.info(
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
);
}
// --- Write Dockerfile with session resume support ---
//
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
// staging COPY/RUN steps. Session staging happens after the original CMD —
// COPY and RUN are layer ops independent of CMD, so the original
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
//
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
const taskSharedDockerfile = existsSync(perMemberDockerfile)
? perMemberDockerfile
: join(repoRoot, 'task-shared', 'Dockerfile');
let baseDockerfile: string;
if (existsSync(taskSharedDockerfile)) {
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
} else {
log.warn(
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
);
baseDockerfile = `FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y \\
git \\
python3 \\
curl \\
jq \\
&& rm -rf /var/lib/apt/lists/*
# Install Claude Code globally (needed by the grader in test.sh)
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
WORKDIR /workspace
COPY workspace/ .
# Block network tools — agent should only read code and write documents
RUN mkdir -p .claude && \\
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
RUN git init && \\
git config user.email "dev@agent" && \\
git config user.name "Dev" && \\
git add -A && \\
git commit -m "initial" --quiet
CMD ["sleep", "infinity"]
`;
}
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
// toolkit's own append rather than an edit to the Dockerfile.
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
// directories in the context, so the layer errors with `"/session": not found`.
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
// files are copied into environment/, so the task-side copy is not there yet.
const sessionSiblingDir = join(snapshotDir, 'session');
const hasSessionSibling =
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
const sessionStaging = `
# >>> toolkit-managed: snapshot-session >>>
# Stage session files for the snapshot agent adapter to install at runtime.
COPY session.jsonl /tmp/snapshot-session/session.jsonl
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
# <<< toolkit-managed <<<
`;
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
log.debug('Wrote Dockerfile (per-repo base + session staging)');
// --- Copy snapshot.patch as workspace.patch ---
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
if (existsSync(snapshotPatch)) {
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
log.debug('Copied snapshot.patch -> workspace.patch');
}
// --- Scrub the worker's filesystem layout out of the session ---
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
const WORKSPACE_MOUNT = '/workspace';
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
/** Member names when this toolkit is polyglot; empty means single-repo. */
const MEMBER_NAMES: readonly string[] = (() => {
const dir = join(repoRoot, 'repos');
if (!existsSync(dir)) return [];
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory())
.map((e) => e.name);
} catch {
return [];
}
})();
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
function repoRootOf(cwd: string): string | null {
if (MEMBER_NAMES.length > 0) {
// A real member of THIS toolkit wins; the generic shape covers a member whose
// directory the toolkit no longer has (an older snapshot, a renamed member).
for (const name of MEMBER_NAMES) {
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
if (hit) return hit[1];
}
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
if (generic) return generic[1];
}
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
return m ? m[1] : null;
}
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
// and a session spanning two checkouts is scrubbed rather than skipped.
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
.filter((r): r is string => r !== null)
.sort((a, b) => b.length - a.length);
const { sanitized } = sanitizeSessionJsonl(raw, {
cwdPrefixes: roots,
placeholder: WORKSPACE_MOUNT,
});
return { text: sanitized, roots };
}
// --- Copy session files for --resume ---
//
// The full session.jsonl (including any post-end_turn entries) goes into the
// task root for reference. A truncated version — keeping everything up to
// and including the last assistant entry with stop_reason="end_turn" — goes
// into environment/ for the container. Stopping on a clean assistant turn
// avoids Claude Code's synthetic "No response requested." injection when
// the session is resumed with --fork-session and a new --print prompt.
const sessionJsonl = join(snapshotDir, 'session.jsonl');
if (existsSync(sessionJsonl)) {
// Fail-open: a session this can't scrub ships exactly as it was, because a
// leaked path is a smaller problem than a task that can't be created.
let sessionText = readFileSync(sessionJsonl, 'utf8');
try {
const { text, roots } = scrubWorkerPaths(sessionText);
if (roots.length > 0) {
sessionText = text;
log.info(
{ roots, mountedAt: WORKSPACE_MOUNT },
'Rewrote the authoring checkout path to the trial mount point'
);
} else {
log.debug('No worker-rooted cwd to rewrite; session used as-is');
}
} catch (err) {
log.warn(
{ err: err instanceof Error ? err.message : String(err) },
'Could not rewrite paths in the session; using it as-is'
);
}
// Full version for reference
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
log.debug('Wrote full session.jsonl to task root');
// Truncated version for the container: strip everything from the last
// user text turn onwards. This drops the failure-eliciting question
// (which `--print` will redeliver to the trial agent as the new prompt)
// AND the failure response itself (so the trial agent doesn't see its
// previous answer), while preserving conversational context up to the
// last clean assistant `end_turn`.
//
// Algorithm (refined Option B):
// 1. Find U = index of the last user-text turn that is NOT a slash
// command (use the same command-marker filter as
// extractLastUserMessage).
// 2. Walk backwards from U - 1 to find the last `assistant` entry
// with stop_reason: "end_turn".
// 3. Truncate slice(0, lastEndTurnIndex + 1).
//
// If U doesn't exist or no end_turn assistant precedes U, write an
// empty session.jsonl — the snapshot agent adapter detects this and
// skips --resume entirely, starting fresh from --print.
const sessionLines = sessionText.trimEnd().split('\n');
// A non-Claude session is not a Claude transcript, so the scan below finds no
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
// applies the same rule in that harness's own format.
const harness = metadata.harness ?? 'claude-code';
const isClaude = harness === 'claude-code';
let lastUserTextIndex = -1;
for (let i = 0; i < sessionLines.length; i++) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
// Compaction summaries are synthetic user turns whose text often quotes
// earlier /create-snapshot:snapshot runs — never the command turn itself,
// so they must not trip the break below.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
// Mirror extractLastUserMessage: skip the snapshot command itself
// and any slash-command / local-command marker turns.
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserTextIndex = i;
} catch {
continue;
}
}
let lastEndTurnIndex = -1;
if (lastUserTextIndex > 0) {
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
message?: { stop_reason?: unknown };
};
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
lastEndTurnIndex = i;
break;
}
} catch {
continue;
}
}
}
if (!isClaude) {
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
const truncated = stripAuthoringScaffolding(harness, kept);
writeFileSync(
join(taskDir, 'environment', 'session.jsonl'),
truncated.length ? truncated.join('\n') + '\n' : ''
);
log.debug(
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (harness reader)'
);
} else if (lastEndTurnIndex >= 0) {
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
log.debug(
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
);
} else {
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
if (lastUserTextIndex < 0) {
log.warn(
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
} else {
log.warn(
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
}
}
}
const sessionDir = join(snapshotDir, 'session');
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
// Claude Code writes subagent files write-only (--w-------). Fix them so
// Harbor's dirhash can read them during environment setup.
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
log.debug('Copied session/');
} else {
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
}
// The harness that captured the snapshot; the trial runs this one.
const harness =
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
/**
* The model and effort this harness defaulted to when the task was authored, recorded
* for reference only — nothing reads these back, and a trial still resolves both from
* the registry at run time. Best-effort: a task is not worth failing over a note.
*/
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
try {
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
// registry needs tomllib. Best-effort, so a miss just omits the note.
let python = '';
for (const candidate of [
process.env.RACCOON_PYTHON,
'python3',
'python3.13',
'python3.12',
'python3.11',
]) {
if (!candidate) continue;
try {
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
python = candidate;
break;
} catch {
continue;
}
}
if (!python) return null;
const rows = execFileSync(python, [resolver, '--defaults'], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
});
for (const line of rows.split('\n')) {
const [id, model, effort] = line.split('\t');
if (id === harnessId && model) return { model, effort: effort ?? '' };
}
} catch {
// registry unreadable here — omit the note
}
return null;
}
const authored = authoredDefaults(harness);
// --- Write task.toml ---
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
// so there's nothing to set here.
const taskToml = `version = "1.0"
[metadata]
program = "raccoon"
author = "rl-env-coding"
category = "sdlc/technical-writing"
repo = "${repoName}"
commit = "${commitShort}"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]
timeout_sec = 7200.0
[agent]
harness = "${harness}"
timeout_sec = 18000.0
[environment]
build_timeout_sec = 6000.0
cpus = 2
memory_mb = 4096
storage_mb = 10240
gpus = 0
allow_internet = true
[verifier.env]
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
[solution.env]
`;
writeFileSync(join(taskDir, 'task.toml'), taskToml);
log.debug('Wrote task.toml');
// --- Extract instruction from session transcript ---
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
if (!existsSync(sessionPath)) return null;
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
// and the worker silently gets a placeholder instruction. Its reader applies the same
// rule — last real user turn, ignoring command invocations — in that harness's format.
if (harness !== 'claude-code') {
const userTurns = turnsFromLines(harness, lines).filter(
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
);
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
}
let lastUserMessage: string | null = null;
for (const line of lines) {
try {
const entry = JSON.parse(line) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
// Synthetic compaction summary — not a real user turn, and its text
// often quotes earlier /create-snapshot:snapshot runs.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserMessage = content;
}
} catch {
continue;
}
}
return lastUserMessage;
}
const lastUserMessage = extractLastUserMessage(
join(snapshotDir, 'session.jsonl'),
metadata.harness ?? 'claude-code'
);
const instructionHeader =
'# Replace this with your refined task instruction\n\n' +
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
if (lastUserMessage) {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader + lastUserMessage.trimEnd() + '\n'
);
log.info('Wrote instruction.md (from last user message in session)');
} else {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader +
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
);
log.warn('Could not extract instruction from session — needs manual editing');
}
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
HOLISTIC RUBRIC — the file trials grade against. Run
/write-holistic-rubric
to draft it interactively, or point Claude Code at this file,
session-full.jsonl, and task-shared/grading-standard.md.
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
## What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
## What this file contains
The eight-criterion Grading Standard
(task-shared/grading-standard.md, embedded in
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
Correctness, Broader Correctness / craft, Persistence, Communication,
Verification & Thoroughness, Common Sense, and Thought Partnership. This
file adds the task-specific knowledge the grader cannot infer: full task
context, the ground truth you established, what strong and weak responses
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
fraction subtractions with a named criterion target, never points, never
caps. The document must stand alone: the grader sees only it and the
shared standard.
-->
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
including the instructions above. -->
`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
// --- Build workspace ---
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
if (existsSync(buildScript)) {
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
try {
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
cwd: repoRoot,
encoding: 'utf8',
stdio: 'inherit',
// build-workspace does a bulk-file write burst (git archive|tar of the
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
// falls back to a full copy across filesystems). On a slow bind mount
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
// path) that legitimately runs into minutes, so a tight cap false-fails a
// working-but-slow build as "not runnable". Keep this generous — it's only
// a backstop against a true hang; the real Harbor build downstream budgets
// build_timeout_sec = 6000.
timeout: 1_200_000,
});
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
process.exit(1);
}
} else {
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
process.exit(1);
}
try {
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
log.info('Next steps:');
log.info(' 1. Review instruction.md');
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
log.info(' 3. Run calibration trials to validate scoring tiers');

View File

@@ -0,0 +1,989 @@
"""
Harbor agent adapters built on the stock claude-code adapter.
- ``PreinstalledClaudeCode``: stock behavior, except agent-setup reuses the
claude binary baked into the task image instead of re-downloading it.
- ``SnapshotClaudeCode``: extends it to inject --resume and --fork-session
when a snapshot session is present in the environment.
The harbor-run script uses PreinstalledClaudeCode for manual tasks and
auto-detects snapshot-based tasks to use SnapshotClaudeCode.
"""
import base64
import io
import json
import logging
import os
import shlex
import shutil
import tarfile
import tempfile
from pathlib import Path
import atif_session
import browser_note
try:
from dnsjail import apply_dns_jail
except ImportError: # no helper shipped -> no jail, rather than no trials
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
return None
from harbor.agents.installed.base import CliFlag
from harbor.agents.installed.claude_code import ClaudeCode
from harbor.models.trial.paths import EnvironmentPaths
_UUID_FILE = "/tmp/snapshot-session/uuid.txt"
_SESSION_FILE = "/tmp/snapshot-session/session.jsonl"
_DISABLED_SUFFIX = ".snapshot-seeded-disabled"
# The "[1m]" model-id suffix is a benchmark convention for the 1M-context
# window, NOT a real model id — the wire request must send the plain id plus
# this beta header (the proxy's server-side alias for the suffixed id was
# dropped; the literal id now 400s and the claude CLI hangs retrying).
_CONTEXT_1M_BETA_HEADER = "anthropic-beta: context-1m-2025-08-07"
def _strip_1m_suffix(model: str) -> tuple[str, bool]:
"""Split a model id into (wire id, wants-1M-context). The [1m] tag stays in
agent identity (name(), trial-config model_name); only the wire id drops it."""
if model.endswith("[1m]"):
return model[: -len("[1m]")], True
return model, False
def _merge_custom_headers(*values: str | None) -> str:
"""Union newline-separated ANTHROPIC_CUSTOM_HEADERS values, keeping order.
Local, not shared: this module ships to the toolkit without llm_proxy_env."""
merged: list[str] = []
for value in values:
for line in (value or "").split("\n"):
if line.strip() and line not in merged:
merged.append(line)
return "\n".join(merged)
def _with_1m_beta_header(existing: str | None) -> str:
"""Merge the 1M-context beta header into an ANTHROPIC_CUSTOM_HEADERS value."""
return _merge_custom_headers(existing, _CONTEXT_1M_BETA_HEADER)
# Fast mode (claude's /fast), opted in per-trial via `--ak fast_mode=true`
# (harbor-run --fast). The --settings flag is the only headless opt-in, and it
# doubles as the availability override for a proxy base URL: claude probes
# org fast-mode status at an API route inference-only proxies don't forward,
# and without the flag that failed probe reads as "unavailable". Rendered
# pre-quoted because a bool CliFlag emits its `cli` string verbatim into a
# shell command.
_FAST_MODE_CLI = "--settings " + shlex.quote('{"fastMode":true}')
_log = logging.getLogger("snapshot-agent")
# --- Reduced "bash + str_replace_editor" tool surface (the CANONICAL agent) --
# The canonical agent runs with ONLY the built-in Bash tool (so there's no
# async-MCP startup race), and a str_replace_editor file editor is delivered as a
# CLI it invokes through Bash. The editor's logic is vendored verbatim under
# scripts/str_replace_editor_vendor/ and wrapped by scripts/str_replace_editor.
# We stage both into the sandbox at install time and point the agent at them via
# --append-system-prompt.
#
# The toolset is chosen by WHICH AGENT CLASS harbor runs, not by an env var: the
# reduced toolset is the canonical PreinstalledClaudeCode / SnapshotClaudeCode;
# the full Claude Code built-in toolset (Read/Edit/Write/Grep/...) is the SEPARATE,
# transitional FullToolsetPreinstalledClaudeCode / FullToolsetSnapshotClaudeCode
# (delete those once every snapshot session.jsonl is recorded in the reduced
# format). A different toolset is simply a different agent — see name() below.
_AGENT_CLI_DIR = "/opt/agent-cli"
_AGENT_CLI_BIN = f"{_AGENT_CLI_DIR}/str_replace_editor"
# Files copied (orchestrator-relative) into the sandbox tar, arcname -> source.
_AGENT_CLI_FILES = {
"str_replace_editor_vendor/__init__.py": "str_replace_editor_vendor/__init__.py",
"str_replace_editor_vendor/base.py": "str_replace_editor_vendor/base.py",
"str_replace_editor_vendor/run.py": "str_replace_editor_vendor/run.py",
"str_replace_editor_vendor/edit.py": "str_replace_editor_vendor/edit.py",
"str_replace_editor": "str_replace_editor",
}
_AGENT_CLI_NOTE_FALLBACK = (
"You are running with a restricted toolset: your ONLY built-in tool is Bash.\n\n"
"To view and edit files, use the `str_replace_editor` command-line tool (it "
"replicates the standard str_replace-based file editor). Invoke it from Bash "
f"by piping ONE JSON object to {_AGENT_CLI_BIN} on stdin. Use a quoted "
"heredoc so backslashes and quotes are preserved:\n\n"
f" {_AGENT_CLI_BIN} <<'EDITOR'\n"
' {"command":"view","path":"/abs/path/file.rb"}\n'
" EDITOR\n\n"
"Commands (the JSON \"command\" field):\n"
"- view: view a file (optionally add \"view_range\":[start,end]) or list a directory.\n"
"- create: create a NEW file -> {\"command\":\"create\",\"path\":...,\"file_text\":\"...\"} (fails if it already exists).\n"
"- str_replace: replace a UNIQUE substring -> {\"command\":\"str_replace\",\"path\":...,\"old_str\":\"...\",\"new_str\":\"...\"}.\n"
"- insert: insert at a line -> {\"command\":\"insert\",\"path\":...,\"insert_line\":N,\"insert_text\":\"...\"}.\n\n"
"All paths must be absolute. JSON strings must be valid (escape newlines as \\n "
"and double-quotes as \\\"). For everything else (running commands, searching "
"with grep/find, reading via sed, etc.) use Bash directly."
)
def _toolset_note(with_browser: bool = False, with_read: bool = False) -> str:
"""The toolset note appended to Claude Code's stock ``--print`` system prompt
(via ``--append-system-prompt``) for the canonical reduced toolset.
We APPEND rather than replace: the stock prompt's big block carries the
DYNAMIC Environment section (cwd, platform, OS, model) generated per run, and
``--system-prompt`` (full replace) would drop it — leaving the reduced-toolset
agent without the Environment/Memory sections the full-toolset agent has (and
those differ host-vs-sandbox, so they can't be hardcoded faithfully).
Appending keeps the sandbox's real block intact; this note is added last to
override the two stock spots that name tools this harness lacks (the "prefer
the dedicated file/search tools" Harness bullet and the Memory section's "use
the Write tool"). Single source of truth is toolset_note.md (read as-is, with
only surrounding whitespace trimmed). Falls back to the built-in note if the
file is missing."""
path = Path(__file__).resolve().parent / "toolset_note.md"
try:
note = path.read_text(encoding="utf-8").strip()
except OSError:
note = _AGENT_CLI_NOTE_FALLBACK
# Order matters: the Read correction must come AFTER the base note, because it supersedes
# that note's "there are no Read/Grep/Glob tools" line. Shipping the base note alone to a
# Read-enabled agent would be a false statement about its own toolset.
if with_read:
read_note = Path(__file__).resolve().parent / "toolset_note_read.md"
try:
note = f"{note}\n\n{read_note.read_text(encoding='utf-8').strip()}"
except OSError:
_log.warning("toolset_note_read.md missing; Read correction omitted")
if with_browser:
extra = browser_note.browser_note()
if extra:
note = f"{note}\n\n{extra}"
return note
class PreinstalledClaudeCode(ClaudeCode):
"""Canonical agent: the reduced ``bash + str_replace_editor`` toolset, with
agent-setup reusing the claude binary baked into the task image instead of
re-downloading it. Manual (non-snapshot) tasks use this directly.
A different toolset is a different AGENT (not an env-var flag), so this class
names itself ``claude-code-reduced-toolset`` — the agent identity carries the
toolset and there's no separate toolset field anywhere. The full Claude Code
built-in toolset is the transitional :class:`FullToolsetPreinstalledClaudeCode`
(name ``claude-code``). (Harbor records the name in each trial's result.json;
we don't use ``harbor traces export`` — which would otherwise require a
registry name — so a descriptive non-registry name is fine. Provenance is also
in the trial config's ``agent.import_path``.)
"""
# Set by _probe_browser() during install(); read by build_cli_flags(). Declared
# here so the full-toolset subclass (which skips the probe) still has a value.
_has_browser = False
# Adding fast_mode HERE (the shared base) makes `--ak fast_mode=true` a known
# kwarg on every Claude agent class, and every build_cli_flags variant —
# reduced, browser, full-toolset — renders it via this table.
CLI_FLAGS = [
*ClaudeCode.CLI_FLAGS,
CliFlag("fast_mode", cli=_FAST_MODE_CLI, type="bool"),
]
@staticmethod
def name() -> str:
return "claude-code-reduced-toolset"
# Manual tasks inherit harbor's run(), so the jail has to be applied here as well as on
# the snapshot path — which overrides run() and never reaches this one.
async def run(self, instruction, environment, context) -> None: # type: ignore[override]
await apply_dns_jail(self, environment)
await super().run(instruction, environment, context)
def __init__(self, *args, **kwargs) -> None:
super().__init__(*args, **kwargs)
# Normalize a "[1m]"-suffixed model id HERE, in the shared base of all
# four Claude agent classes, so every run path sends the plain wire id —
# including stock ClaudeCode.run(), which this class inherits for manual
# (non-snapshot) tasks and which reads self.model_name directly. The 1M
# window is requested via the beta header instead, delivered through
# harbor's extra_env channel (merged into every agent exec on 0.9.x,
# wired via Trial.scoped_exec_env on 0.18.x); _build_env also mirrors it
# for the snapshot run path. The [1m] identity survives on purpose:
# BaseAgent's _init_model_info already cached the suffixed id (for
# to_agent_info) before this rebinding, and the trial config's
# agent.model_name records the id as passed on the CLI.
model = self.model_name
if not model:
# Stock ClaudeCode.run() falls back to os.environ["ANTHROPIC_MODEL"]
# verbatim, with no overridable hook on the manual-task path — so
# when no model was pinned, adopt a [1m]-suffixed env model here
# (same effective wire value, normalized). A plain env model stays
# on the stock fallback path untouched.
model = os.environ.get("ANTHROPIC_MODEL", "")
stripped, wants_1m = _strip_1m_suffix(model)
if wants_1m:
self.model_name = stripped
self._extra_env["ANTHROPIC_CUSTOM_HEADERS"] = _with_1m_beta_header(
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS")
)
async def install(self, environment) -> None:
"""Canonical (reduced) setup: stage the str_replace_editor CLI, then ensure
the claude binary. The full-toolset subclass skips the staging (it has no
editor CLI) and reuses ``_ensure_claude_binary`` directly."""
await self._stage_agent_cli(environment)
await self._ensure_claude_binary(environment)
await self._probe_browser(environment)
async def _probe_browser(self, environment) -> None:
"""Record whether this image ships the Playwright `pw` wrapper, so the toolset
note mentions the browser only on images that have one. Runs during install(),
which harbor calls before build_cli_flags() reads the result. The probe and the
note text are shared with the codex adapter via browser_note.py."""
self._has_browser = await browser_note.probe_browser(environment)
async def _ensure_claude_binary(self, environment) -> None:
"""Reuse the claude binary already baked into the task image instead
of re-downloading it at agent-setup.
Harbor's stock ``install()`` pipes ``claude.ai/install.sh`` to bash
inside the live sandbox — a ~240 MB binary download racing the 360 s
agent-setup timeout. On a slow or stalling egress path (classic on
WSL2/Docker Desktop) the download never finishes and every trial dies
with ``AgentSetupTimeoutError`` — pure waste, since our task
Dockerfiles already bake claude into ``/usr/local/bin``. Probe for a
working binary first; fall back to harbor's installer only when the
image truly lacks one, when an explicit agent-version pin doesn't
match the baked binary, or when the probe itself errors. The probe
is local and sub-second, so the fallbacks are effectively no worse
than stock behavior.
"""
try:
probe = await environment.exec(
command=(
'export PATH="$HOME/.local/bin:$PATH"; '
"command -v claude >/dev/null 2>&1 && claude --version"
),
timeout_sec=30,
)
except Exception as exc: # noqa: BLE001 — any probe failure → stock path
_log.debug(
"claude preinstall probe failed (%s); using stock installer", exc
)
await ClaudeCode.install(self, environment)
return
if probe.return_code == 0:
pinned = getattr(self, "_version", None)
baked = self.parse_version(probe.stdout or "")
if not pinned or baked == pinned:
_log.info(
"claude already in image (%s); skipping runtime download",
(probe.stdout or "").strip(),
)
return
_log.info(
"image bakes claude %s but %s was requested; using stock installer",
baked,
pinned,
)
await ClaudeCode.install(self, environment)
def build_cli_flags(self) -> str:
"""Emit the reduced ``bash + str_replace_editor`` toolset flags: restrict
the built-in toolset to ``--tools Bash`` and append the toolset note.
harbor's stock adapter exposes only ``--allowedTools`` /
``--disallowedTools`` (permission lists). Under
``--permission-mode=bypassPermissions`` (which both run paths use) an
allowlist does NOT remove tools — every built-in stays available, just
auto-approved. claude's ``--tools`` flag is the one that sets the
AVAILABLE toolset; ``Bash`` leaves Bash as the only built-in.
APPEND (not replace) the toolset note: --append keeps the sandbox's real,
dynamically-generated Environment/Memory block intact, and the note (added
last) overrides the stock prompt's references to tools this harness lacks
(the "prefer the dedicated file/search tools" bullet and the Memory
section's "use the Write tool"). See _toolset_note(). The full-toolset
subclass overrides this back to stock ``ClaudeCode.build_cli_flags``.
"""
flags = super().build_cli_flags()
note = _toolset_note(with_browser=self._has_browser)
extra = f"--tools Bash --append-system-prompt {shlex.quote(note)}"
return f"{flags} {extra}" if flags else extra
async def _claude_format_session_path(self, environment, env, session_uuid: str) -> str:
"""Path in the sandbox to a session.jsonl Claude can resume.
A task authored on another harness stages THAT harness's native blob, which
`claude --resume` cannot read. Convert it through the ATIF hub and upload the
result. A Claude-authored task — every task before multi-harness authoring —
returns the staged path untouched, so its install stays byte-identical.
"""
async def _read(command: str) -> str | None:
try:
result = await environment.exec(command=command, env=env, timeout_sec=30)
except Exception as exc: # best-effort: fall back to installing as-is
_log.warning("Could not read staged session (%s); installing verbatim", exc)
return None
return getattr(result, "stdout", "") or ""
# Probe the head first: a Claude-authored session needs no conversion, and that is
# the common case, so pulling a multi-megabyte transcript through exec to learn
# only that is waste. Act on the probe only when it says "claude" — a truncated
# ATIF blob (one JSON object) parses as nothing, which is not the same answer.
head = await _read(f"head -c 8192 {_SESSION_FILE} 2>/dev/null || true")
if head is None:
return _SESSION_FILE
if atif_session.detect_format(head) == "claude":
return _SESSION_FILE
text = await _read(f"cat {_SESSION_FILE} 2>/dev/null || true")
if text is None:
return _SESSION_FILE
fmt = atif_session.detect_format(text)
if fmt in (None, "claude"):
return _SESSION_FILE
from datetime import datetime, timezone
iso_ts = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z")
lines = atif_session.atif_to_claude_session(
atif_session.to_atif(text), session_id=session_uuid, iso_ts=iso_ts
)
if not lines:
_log.warning("Converted %s session was empty; installing verbatim", fmt)
return _SESSION_FILE
_log.info(
"Seeding Claude from a %s-authored session: %d records via ATIF", fmt, len(lines)
)
remote = "/tmp/snapshot-session/session.claude.jsonl"
with tempfile.NamedTemporaryFile(
"w", suffix=".jsonl", delete=False, encoding="utf-8"
) as tmp:
tmp.write("\n".join(lines) + "\n")
host_path = tmp.name
try:
await environment.upload_file(host_path, remote)
if environment.default_user is not None:
await self.exec_as_root(
environment, command=f"chown {environment.default_user} {shlex.quote(remote)}"
)
finally:
try:
os.unlink(host_path)
except OSError:
pass
return remote
async def _stage_agent_cli(self, environment) -> None:
"""Stage the vendored str_replace_editor CLI into the sandbox.
Bundles scripts/str_replace_editor_vendor/ (the verbatim EditTool source)
plus the wrapper into a tar, ships it as base64 in one root exec, and
unpacks it to ``/opt/agent-cli`` with the wrapper made executable. Runs at
install time so the CLI is present before the agent's first turn.
"""
here = Path(__file__).resolve().parent
buf = io.BytesIO()
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
for arcname, rel in _AGENT_CLI_FILES.items():
src = here / rel
if not src.is_file():
raise FileNotFoundError(
f"str_replace_editor: missing vendored file {src} "
f"(expected scripts/{rel})"
)
tar.add(str(src), arcname=arcname)
b64 = base64.b64encode(buf.getvalue()).decode()
command = (
f"mkdir -p {_AGENT_CLI_DIR} && "
f"printf %s {shlex.quote(b64)} | base64 -d | "
f"tar xzf - -C {_AGENT_CLI_DIR} && "
f"chmod -R a+rX {_AGENT_CLI_DIR} && chmod a+rx {_AGENT_CLI_BIN} && "
# Fail loudly at setup if python3 is absent — the CLI needs it.
f"command -v python3 >/dev/null 2>&1 || "
f'{{ echo "str_replace_editor: python3 not found in sandbox" >&2; exit 1; }}'
)
await self.exec_as_root(environment, command=command, timeout_sec=120)
_log.info("Staged str_replace_editor CLI to %s", _AGENT_CLI_BIN)
await self._selftest_agent_cli(environment)
async def _selftest_agent_cli(self, environment) -> None:
"""Fail loudly at setup if the staged CLI can't actually run an edit.
Invokes the REAL staged wrapper (``_AGENT_CLI_BIN``) exactly the way the
agent will — one JSON object piped on stdin — and asserts the edit landed
on disk. This exercises the whole path end-to-end (the wrapper's shebang,
its exec permissions, stdin JSON parsing, ``sys.path`` into the vendored
package, and the vendored EditTool itself), so a broken wrapper, wrong
perms, or a bad python (the vendored edit.py uses PEP 604 ``X | None`` and
needs python >= 3.10) errors the trial at agent-setup instead of silently
mid-benchmark. The python version is logged for visibility.
``set -e`` + the trailing OK echo mean any failure in the pipe (the
wrapper) or the grep yields rc != 0 / no OK marker, which we raise on."""
json_fmt = (
'{"command":"str_replace","path":"%s",'
'"old_str":"alpha","new_str":"ALPHA"}'
)
cmd = (
"python3 --version 2>&1; "
"set -e; "
'TMP="$(mktemp)"; '
'printf "alpha\\nbeta\\n" > "$TMP"; '
f"printf '{json_fmt}' \"$TMP\" | {_AGENT_CLI_BIN}; "
'grep -q ALPHA "$TMP"; '
'echo "str_replace_editor self-test OK"'
)
result = await self.exec_as_root(environment, command=cmd, timeout_sec=30)
out = (getattr(result, "stdout", "") or "").strip()
rc = getattr(result, "return_code", 0)
_log.info("str_replace_editor self-test (rc=%s): %s", rc, out.replace("\n", " | "))
if rc != 0 or "self-test OK" not in out:
raise RuntimeError(
f"str_replace_editor self-test failed (rc={rc}). The staged "
f"str_replace_editor could not perform an edit in the sandbox "
f"(often python < 3.10). Output:\n{out}"
)
class SnapshotClaudeCode(PreinstalledClaudeCode):
"""Canonical snapshot agent: resumes from a snapshot session and runs the
reduced ``bash + str_replace_editor`` toolset. The full Claude Code built-in
toolset is the transitional :class:`FullToolsetSnapshotClaudeCode`."""
@staticmethod
def name() -> str:
return "snapshot-claude-code-reduced-toolset"
@staticmethod
def _is_bedrock_mode() -> bool:
return False
async def run(self, instruction: str, environment, context) -> None:
env = self._build_env()
config_dir = env["CLAUDE_CONFIG_DIR"]
# Read the snapshot session UUID from the container
result = await environment.exec(
command=f"cat {_UUID_FILE} 2>/dev/null || echo ''",
env=env,
timeout_sec=5,
)
session_uuid = result.stdout.strip() if result.stdout else ""
# Check whether the staged session.jsonl has any meaningful content.
# snapshot-to-task may write an empty file when the snapshot has no
# assistant entry with stop_reason="end_turn" — in that case we must
# NOT pass --resume / --fork-session (CC errors out on an empty
# session) and we must NOT stage the empty file.
session_has_content = False
if session_uuid:
size_result = await environment.exec(
command=(
f"if [ -s {_SESSION_FILE} ] && "
f"grep -q '[^[:space:]]' {_SESSION_FILE} 2>/dev/null; "
f"then echo 'yes'; else echo 'no'; fi"
),
env=env,
timeout_sec=5,
)
session_has_content = (
size_result.stdout.strip() == "yes" if size_result.stdout else False
)
install_src = _SESSION_FILE
escaped_instruction = shlex.quote(instruction)
cli_flags = self.build_cli_flags()
extra_flags = (cli_flags + " ") if cli_flags else ""
# Install the session JSONL from the container's filesystem (COPY'd
# in by the Dockerfile) into Claude Code's config dir. This runs in
# the same exec call as the claude command so files are visible.
workspace_dir = f"{config_dir}/projects/-workspace"
if session_uuid and session_has_content:
resume_flags = f"--resume {session_uuid} --fork-session "
seeded_jsonl = f"{workspace_dir}/{session_uuid}.jsonl"
install_src = await self._claude_format_session_path(
environment, env, session_uuid
)
install_prefix = (
f'mkdir -p "{workspace_dir}" && '
f'cp "{install_src}" "{seeded_jsonl}" && '
f'chmod -R 777 "{config_dir}" && '
)
# After the run, drop the seeded session JSONL so harbor's
# trajectory converter sees ONLY the forked session claude wrote.
# ``--fork-session`` writes the forked conversation (a SUPERSET: it
# copies the seeded history verbatim, reusing the same ``toolu_*``
# ids) to a NEW ``{uuid}.jsonl``. If the seed is left behind,
# harbor's ``_convert_events_to_trajectory`` globs BOTH files and
# merges them, producing duplicate ``tool_result`` events; the
# second one orphans a tool_call (empty ``tool_name``), which
# ``_convert_event_to_step`` skips, leaving a gap that trips the
# sequential ``step_id`` invariant on the ``Trajectory`` model — so
# the whole conversion raises and ``trajectory.json`` never lands.
# Removing the seed in-sandbox makes a snapshot trial look exactly
# like a stock claude-code trial (one session file) and is robust
# even when the host-side ``populate_context_post_run`` hook below
# is bypassed (e.g. a stale agent module on harbor's import path).
# Guard: only remove the seed if a *different* forked JSONL exists,
# so a claude build that appended in place (no real fork) keeps its
# sole session file.
seeded_cleanup = "; " + self._seeded_cleanup_cmd(
workspace_dir, session_uuid
)
else:
resume_flags = ""
install_prefix = ""
seeded_cleanup = ""
if session_uuid and not session_has_content:
# Seeded session.jsonl is empty (no end_turn assistant in the
# source snapshot). Start fresh with --print instead.
_log.debug(
"Seeded session.jsonl is empty; skipping --resume and starting fresh"
)
await apply_dns_jail(self, environment)
await self.exec_as_agent(
environment,
command=(
f'{install_prefix}'
f'export PATH="$HOME/.local/bin:$PATH"; '
f'export CLAUDE_CONFIG_DIR="{config_dir}"; '
f"claude --verbose --output-format=stream-json "
f"--permission-mode=bypassPermissions "
f"{resume_flags}"
f"{extra_flags}"
f"--print -- {escaped_instruction} 2>&1 </dev/null | tee "
f"/logs/agent/claude-code.txt"
f"{seeded_cleanup}"
),
env=env,
)
def _build_env(self) -> dict[str, str]:
"""Build the environment dict for agent execution."""
env: dict[str, str | None] = {
"ANTHROPIC_API_KEY": os.environ.get("ANTHROPIC_API_KEY", ""),
"ANTHROPIC_BASE_URL": os.environ.get("ANTHROPIC_BASE_URL", None),
"CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
"IS_SANDBOX": "1",
"FORCE_AUTO_BACKGROUND_TASKS": "1",
"ENABLE_BACKGROUND_TASKS": "1",
}
# The host env's header list (e.g. the proxy's X-Project-Id entry from
# apply_llm_proxy_env) rides along with any agent-requested headers —
# but only on a proxy base URL: a project id must never reach a
# provider's own API (the direct-API escape hatch).
on_proxy = "/llm_proxy/" in (env["ANTHROPIC_BASE_URL"] or "")
header = _merge_custom_headers(
os.environ.get("ANTHROPIC_CUSTOM_HEADERS") if on_proxy else None,
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS"),
)
if self.model_name:
# "[1m]" normalization happened once in PreinstalledClaudeCode.__init__
# (shared by all Claude agent classes); by here model_name is the plain
# wire id and any 1M beta header sits in self._extra_env.
env["ANTHROPIC_MODEL"] = self.model_name.split("/")[-1]
elif "ANTHROPIC_MODEL" in os.environ:
# A [1m] env model was adopted into model_name by __init__, so this
# fallback normally only sees plain ids — but the env can change
# after construction, so normalize here too (defense in depth).
fallback, wants_1m = _strip_1m_suffix(os.environ["ANTHROPIC_MODEL"])
if wants_1m:
header = _with_1m_beta_header(header)
env["ANTHROPIC_MODEL"] = fallback
# Mirror the [1m] beta header into this env dict: harbor versions differ
# in where extra_env is merged (0.9.x: per-exec in _exec; 0.18.x:
# Trial-level scoped_exec_env), so carrying it here keeps the snapshot
# run path correct regardless of which plumbing the installed harbor has.
if header:
env["ANTHROPIC_CUSTOM_HEADERS"] = header
if os.environ.get("CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING", "").strip() == "1":
env["CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING"] = "1"
env.update(self._resolved_env_vars)
env["CLAUDE_CONFIG_DIR"] = (EnvironmentPaths.agent_dir / "sessions").as_posix()
return {k: v for k, v in env.items() if v}
@staticmethod
def _seeded_cleanup_cmd(workspace_dir: str, session_uuid: str) -> str:
"""Shell snippet (run after the claude pipeline) that deletes the seeded
session JSONL so harbor's converter sees ONLY the forked session.
Guarded: the seed is removed only if a *different* ``*.jsonl`` exists in
``workspace_dir`` — i.e. claude actually forked to a new file. If claude
appended in place (no real fork), the seed is the sole session file and
is kept, so we never destroy the only record of the run.
"""
seeded_jsonl = f"{workspace_dir}/{session_uuid}.jsonl"
return (
f'if [ -f "{seeded_jsonl}" ] && '
f'ls "{workspace_dir}/"*.jsonl 2>/dev/null '
f'| grep -vq "/{session_uuid}\\.jsonl$"; '
f'then rm -f "{seeded_jsonl}"; fi'
)
def populate_context_post_run(self, context) -> None:
"""Override harbor's post-run hook to make trajectory.json production
reliable for snapshot resumes.
The seeded session JSONL (the file we cp'd in from
``/tmp/snapshot-session/session.jsonl`` during run-prep) lives in
``sessions/projects/-workspace/{seeded_uuid}.jsonl`` alongside the
new forked-session JSONL claude actually wrote during this trial.
Harbor's ``_convert_events_to_trajectory`` reads BOTH files, and
events from the seeded session — often from an older claude version
with a slightly different schema — trip skip-paths inside
``_convert_event_to_step``. Skipping events breaks the sequential
``step_id`` invariant the ``Trajectory`` pydantic model enforces, so
the whole conversion errors out and ``trajectory.json`` never lands.
We work around it by moving every JSONL whose stem is not the
forked session id aside before delegating to the parent's hook,
then restoring it after. The forked session id is read from
``claude-code.txt``'s ``system/init`` event, which is the first
thing claude writes via ``--output-format=stream-json``.
If, after our cleanup, trajectory.json still isn't there, we log
at WARNING (not debug) so downstream consumers can see something
went wrong rather than silently inherit a half-broken run.
"""
moved = self._isolate_forked_session_jsonl()
try:
super().populate_context_post_run(context)
finally:
self._restore_moved_jsonls(moved)
trajectory_path = self.logs_dir / "trajectory.json"
if not trajectory_path.is_file():
# Stock conversion produced nothing even from the isolated forked
# session. The usual culprit is a single orphaned tool_result (e.g.
# left behind by autocompaction) that harbor skips, leaving a
# step_id gap that sinks the whole Trajectory. Recover by rebuilding
# from the forked session with orphaned tool_results stripped — one
# degenerate event should not cost us the entire trajectory.
if self._recover_trajectory_stripping_orphans():
self.logger.info(
"Recovered trajectory.json by stripping orphaned tool_results"
)
if not trajectory_path.is_file():
self.logger.warning(
"trajectory.json was NOT produced in %s. Downstream tooling "
"(grader replay, worldbench export, publish-reference-runs) "
"depends on it. claude-code.txt: %s. sessions/projects: %s",
self.logs_dir,
"present" if (self.logs_dir / "claude-code.txt").is_file() else "MISSING",
self._summarize_sessions_dir(),
)
def _read_forked_session_id(self) -> str | None:
"""Return the ``session_id`` from the first ``system/init`` event in
``claude-code.txt``, or None if it can't be found.
Claude Code's ``--output-format=stream-json --print`` mode emits the
init event near the top of the stream, but is allowed to emit other
framing events before it (provider notices, warnings, etc.). Scan
every line until we find an init event with a usable session_id, or
we hit EOF — don't bail on the first non-init JSON we see.
"""
stream_path = self.logs_dir / "claude-code.txt"
if not stream_path.is_file():
return None
try:
with open(stream_path, "r", encoding="utf-8") as handle:
for line in handle:
stripped = line.strip()
if not stripped or not stripped.startswith("{"):
continue
try:
event = json.loads(stripped)
except json.JSONDecodeError:
continue
if (
event.get("type") == "system"
and event.get("subtype") == "init"
):
sid = event.get("session_id")
if isinstance(sid, str) and sid:
return sid
except OSError:
return None
return None
def _recover_trajectory_stripping_orphans(self) -> bool:
"""Last-resort rebuild of ``trajectory.json`` from the forked session,
tolerant of the single degenerate event that harbor's converter would
otherwise let sink the whole trajectory.
Harbor assigns ``step_id`` from the enumerate index BEFORE it may skip an
event, so any event that ``_convert_event_to_step`` raises on leaves a
gap that fails ``Trajectory.validate_step_ids`` — and the whole
conversion is lost. Two real shapes trigger this even in a single,
already-isolated forked session:
- an orphaned ``tool_result`` whose ``tool_use`` was summarized away by
autocompaction; and
- a ``tool_result`` that shares an identical timestamp with its
``tool_use`` and stable-sorts ahead of it, so the result is processed
before the call exists (also yielding an empty ``tool_name``).
We re-run harbor's own conversion but temporarily make
``_convert_event_to_step`` substitute a placeholder step instead of
raising, so one bad event costs us that single observation rather than
the entire run. Returns ``True`` iff ``trajectory.json`` was written.
This runs ONLY after the stock conversion already failed, so it never
changes behavior on healthy sessions.
"""
sessions_root = self.logs_dir / "sessions" / "projects"
if not sessions_root.is_dir():
return False
jsonls: list[Path] = []
for project_dir in sessions_root.iterdir():
if project_dir.is_dir():
jsonls.extend(project_dir.glob("*.jsonl"))
if not jsonls:
return False
# Prefer the forked session alone; fall back to whatever is present.
forked = self._read_forked_session_id()
if forked and any(j.stem == forked for j in jsonls):
jsonls = [j for j in jsonls if j.stem == forked]
from harbor.models.trajectories.step import Step
original_convert = self._convert_event_to_step
def tolerant_convert(event: dict, step_id: int) -> Step:
try:
return original_convert(event, step_id)
except ValueError:
# Degenerate tool event (orphaned / mis-ordered tool_result).
# Keep its output as a user observation so nothing is silently
# dropped, and the sequential step_id stays intact.
output = event.get("output")
call_id = event.get("call_id") or "?"
message = (
output
if isinstance(output, str) and output.strip()
else f"[unmatched tool_result for {call_id}]"
)
# Guard the timestamp: Step.validate_timestamp raises ValueError
# on a non-ISO-8601 value, which harbor's loop would catch and
# skip — re-introducing the exact step_id gap we're recovering
# from. Fall back to no timestamp rather than lose the step.
ts = event.get("timestamp")
try:
return Step(
step_id=step_id, timestamp=ts, source="user", message=message
)
except ValueError:
return Step(
step_id=step_id, timestamp=None, source="user", message=message
)
with tempfile.TemporaryDirectory() as tmp:
session_dir = Path(tmp) / "-workspace"
session_dir.mkdir(parents=True)
for jsonl in jsonls:
shutil.copy(jsonl, session_dir / jsonl.name)
self._convert_event_to_step = tolerant_convert # type: ignore[assignment]
try:
trajectory = self._convert_events_to_trajectory(session_dir)
except Exception as exc: # noqa: BLE001
self.logger.debug("Tolerant recovery failed: %s", exc)
return False
finally:
del self._convert_event_to_step
if not trajectory:
return False
try:
with open(self.logs_dir / "trajectory.json", "w", encoding="utf-8") as handle:
json.dump(
trajectory.to_json_dict(), handle, indent=2, ensure_ascii=False
)
except OSError:
return False
return True
def _isolate_forked_session_jsonl(self) -> list[tuple[Path, Path]]:
"""Move any JSONL not matching the forked session id to a sibling
``*.snapshot-seeded-disabled`` path. Returns the list of
(original, disabled) pairs so they can be restored after.
If we can't determine the forked session id (no claude-code.txt, no
init event, etc.), we leave the directory untouched. Harbor's
converter will run as today — if it succeeds, great; if not, our
WARNING fires.
Safety guardrail: if NO JSONL on disk matches the forked id (e.g.
claude wrote to an unexpected path), don't move anything aside —
that would leave the converter with an empty session dir and
guarantee failure. Better to let harbor's normal flow attempt the
conversion against what's actually there.
"""
forked = self._read_forked_session_id()
if not forked:
return []
sessions_root = self.logs_dir / "sessions" / "projects"
if not sessions_root.is_dir():
return []
all_jsonls: list[Path] = []
for project_dir in sessions_root.iterdir():
if not project_dir.is_dir():
continue
all_jsonls.extend(project_dir.glob("*.jsonl"))
has_forked_match = any(j.stem == forked for j in all_jsonls)
if not has_forked_match:
self.logger.warning(
"Forked session id %s from claude-code.txt has no matching "
"JSONL in %s (found: %s). Leaving sessions/ untouched so "
"harbor's converter can attempt against what's there.",
forked,
sessions_root,
[j.name for j in all_jsonls],
)
return []
moved: list[tuple[Path, Path]] = []
for jsonl in all_jsonls:
if jsonl.stem == forked:
continue
disabled = jsonl.with_suffix(jsonl.suffix + _DISABLED_SUFFIX)
try:
shutil.move(str(jsonl), str(disabled))
except OSError as exc:
# All-or-nothing: a partial move would feed the converter a
# MIXED set (forked + still-present seeded), which is the
# exact original failure mode this override exists to prevent.
# Roll back any successful moves and let harbor's converter
# run on the unmodified directory — same outcome as today
# (likely fails, our WARNING fires), no worse.
self.logger.warning(
"Could not move seeded JSONL %s aside: %s. Rolling back "
"any prior moves to avoid feeding the converter a mixed "
"set.",
jsonl,
exc,
)
self._restore_moved_jsonls(moved)
return []
moved.append((jsonl, disabled))
_log.info(
"Moved seeded JSONL %s aside so harbor converter only "
"sees forked session %s",
jsonl.name,
forked,
)
return moved
def _restore_moved_jsonls(self, moved: list[tuple[Path, Path]]) -> None:
for original, disabled in moved:
try:
shutil.move(str(disabled), str(original))
except OSError as exc:
self.logger.warning(
"Could not restore seeded JSONL %s from %s: %s",
original,
disabled,
exc,
)
def _summarize_sessions_dir(self) -> str:
sessions_root = self.logs_dir / "sessions" / "projects"
if not sessions_root.is_dir():
return "missing"
parts: list[str] = []
for project_dir in sorted(sessions_root.iterdir()):
if not project_dir.is_dir():
continue
jsonls = sorted(p.name for p in project_dir.glob("*.jsonl"))
parts.append(f"{project_dir.name}={jsonls}")
return ", ".join(parts) if parts else "no JSONLs"
# --- TRANSITIONAL: full Claude Code built-in toolset -------------------------
# These restore Claude Code's full built-in toolset (Read/Edit/Write/Grep/Glob/
# Task/...) for snapshot session.jsonls recorded in the OLD full-toolset format.
# A different toolset is a different agent, so they keep the original
# "claude-code" / "snapshot-claude-code" names. DELETE this mixin + both classes
# once every snapshot session.jsonl is re-recorded in the reduced format.
class _BrowserToolsetMixin:
"""The canonical reduced toolset PLUS the ``Read`` built-in, for tasks that opt into a
browser: a screenshot is only useful to an agent that can look at it, and ``Read`` is what
turns a PNG on disk into an image the model actually sees.
This is a SEPARATE AGENT, not a flag, per the rule that the toolset is chosen by which class
harbor runs and the class name records it — so a benchmark row can never silently compare an
agent that could see against one that couldn't.
Two things to be clear-eyed about:
* ``Read`` is not image-only. It also reads text files, PDFs and notebooks, so these tasks
get back a file-reading built-in the reduced toolset deliberately removes. There is no
narrower built-in; an image-only MCP tool was rejected because the canonical agent avoids
MCP (see the async-MCP startup race note at the top of this file).
* Tasks on this agent are not comparable with tasks on the canonical one. That is the point
of the distinct name.
"""
def build_cli_flags(self) -> str:
flags = ClaudeCode.build_cli_flags(self)
note = _toolset_note(with_browser=self._has_browser, with_read=True)
extra = f"--tools Bash,Read --append-system-prompt {shlex.quote(note)}"
return f"{flags} {extra}" if flags else extra
class BrowserPreinstalledClaudeCode(_BrowserToolsetMixin, PreinstalledClaudeCode):
"""Reduced toolset + Read, manual (non-snapshot) tasks."""
@staticmethod
def name() -> str:
return "claude-code-reduced-toolset-browser"
class BrowserSnapshotClaudeCode(_BrowserToolsetMixin, SnapshotClaudeCode):
"""Reduced toolset + Read, snapshot tasks."""
@staticmethod
def name() -> str:
return "snapshot-claude-code-reduced-toolset-browser"
class _FullToolsetMixin:
"""Override the canonical reduced toolset back to Claude Code's stock full
built-in toolset: no str_replace_editor CLI to stage, and no --tools / note.
Mixed in BEFORE the reduced base so its install/build_cli_flags win, while the
snapshot resume/fork and the claude-binary probe are still inherited."""
async def install(self, environment) -> None:
# Full toolset has no editor CLI to stage — just ensure the claude binary.
await self._ensure_claude_binary(environment)
def build_cli_flags(self) -> str:
# Stock Claude Code flags: the full built-in toolset, no --tools, no note.
return ClaudeCode.build_cli_flags(self)
class FullToolsetPreinstalledClaudeCode(_FullToolsetMixin, PreinstalledClaudeCode):
"""TRANSITIONAL full-toolset manual agent. Delete once snapshots are reduced-format."""
@staticmethod
def name() -> str:
return "claude-code"
class FullToolsetSnapshotClaudeCode(_FullToolsetMixin, SnapshotClaudeCode):
"""TRANSITIONAL full-toolset snapshot agent. Delete once snapshots are reduced-format."""
@staticmethod
def name() -> str:
return "snapshot-claude-code"

View File

@@ -0,0 +1,295 @@
/**
* stage-atomic-rubric.ts — stage a task's atomic rubric into the grading
* copies the rubric grader modes read, so
* `HARBOR_GRADER_MODE=rubric-trinary scripts/harbor-regrade …` can run inside
* this toolkit.
*
* Source of truth: the task's own atomic rubric —
* `tests/atomic-rubric.yaml` (the current name), or `tests/rubrics.yaml` on a
* task converted before the rename. A task carrying BOTH names with different
* content is a hard error: silently preferring either file could stage a
* rubric that is not the one just edited, and the grader would grade the
* wrong criteria with no signal.
*
* Staged into `harbor-tasks/<slug>/tests/`:
* - `rubric-criteria.md` — the criteria text shown to the grader: one
* "### Criterion: <id>" section per criterion, guideline and elaboration
* only. Category, severity, and dimensions are stripped, so the grader
* stays severity-blind.
* - `rubric-criteria.json` — {task, criteria: [{id, category, severity,
* dimensions}]} for `render-rubric-grade.py` (criterion-id validation and
* severity-weighted aggregation). The grader never sees this file.
* - `render-rubric-grade.py` — synced from `task-shared/` when the task's
* copy is missing or differs from the shared source.
*
* `tests/grader-context.md` is part of the task's own package — this script
* checks that it exists and never writes it. Write it alongside the rubric;
* the `/write-atomic-rubric` skill covers both files.
*
* Staged files are derived from the rubric. Re-run this script after every
* rubric edit, and run `--restore` to remove the staged copies. This script
* validates structure only (readable YAML, unique criterion ids, at most two
* Crux criteria); the `/detector-rubric-coverage` and `/detector-rubric-form`
* skills are the content review.
*
* Usage:
* npx tsx scripts/stage-atomic-rubric.ts <task-slug>
* npx tsx scripts/stage-atomic-rubric.ts <task-slug> --restore
*/
import { createHash } from 'node:crypto';
import { copyFileSync, existsSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
import { createRequire } from 'node:module';
import { basename, dirname, isAbsolute, join, relative, resolve } from 'node:path';
import { fileURLToPath } from 'node:url';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
const __dirname = dirname(fileURLToPath(import.meta.url));
const TOOLKIT_ROOT = resolve(__dirname, '..');
const SHARED_DIR = join(TOOLKIT_ROOT, 'task-shared');
const argv = yargs(hideBin(process.argv))
.usage('Usage: $0 <task> [options]')
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
.option('restore', {
type: 'boolean',
default: false,
describe: 'Remove the files a previous staging created',
})
.option('json', { type: 'boolean', default: false, describe: 'Structured JSON logs' })
.demandCommand(1, 'Name the task to stage: npx tsx scripts/stage-atomic-rubric.ts <task-slug>')
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'stage-atomic-rubric', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
/** The atomic-rubric filenames, current name first. */
const RUBRIC_NAMES = ['atomic-rubric.yaml', 'rubrics.yaml'] as const;
/** Record of exactly what staging created, so --restore removes only that. */
const STAGE_MANIFEST = '.rubric-staged.json';
/** The rubric shape this script needs. Validation is structural only — the
* rubric detectors and the repo-side validator own the content rules. */
interface RubricCriterion {
readonly id: string;
readonly category: string;
readonly severity?: string | null;
readonly guideline: string;
readonly elaboration?: string | null;
readonly dimensions?: readonly string[];
}
interface RubricsDoc {
readonly task: string;
readonly criteria: readonly RubricCriterion[];
}
function fail(message: string): never {
log.error(message);
process.exit(1);
}
/** harbor-tasks/<slug> from a slug or a path, mirroring check-task-infra.ts. */
function resolveTaskDir(task: string): string {
const candidate = isAbsolute(task) ? task : resolve(process.cwd(), task);
if (existsSync(join(candidate, 'task.toml'))) return candidate;
const bySlug = join(TOOLKIT_ROOT, 'harbor-tasks', task);
if (existsSync(join(bySlug, 'task.toml'))) return bySlug;
return fail(`No task found at ${task} or harbor-tasks/${task} (expected a task.toml inside).`);
}
const sha256 = (p: string): string => createHash('sha256').update(readFileSync(p)).digest('hex');
/** The task's rubric file — current name first, pre-rename name honored, both
* present with different content refused. */
function resolveRubricPath(testsDir: string): string {
const present = RUBRIC_NAMES.map((name) => join(testsDir, name)).filter((p) => existsSync(p));
if (present.length === 0) {
return fail(
`No atomic rubric found: expected tests/atomic-rubric.yaml ` +
`(or tests/rubrics.yaml on a task converted before the rename). ` +
`Write it with the /write-atomic-rubric skill first.`
);
}
if (present.length > 1 && new Set(present.map(sha256)).size > 1) {
return fail(
`Both tests/atomic-rubric.yaml and tests/rubrics.yaml exist with different content. ` +
`Keep exactly one; tests/atomic-rubric.yaml is the current name.`
);
}
return present[0];
}
/** Structural gate: the properties the staged outputs are built from. */
function toRubricsDoc(raw: unknown, sourceName: string): RubricsDoc {
if (typeof raw !== 'object' || raw === null || Array.isArray(raw)) {
return fail(`${sourceName} is not a YAML mapping with task and criteria keys.`);
}
const doc = raw as { task?: unknown; criteria?: unknown };
if (typeof doc.task !== 'string' || doc.task.length === 0) {
return fail(`${sourceName} is missing the top-level task key.`);
}
if (!Array.isArray(doc.criteria) || doc.criteria.length === 0) {
return fail(`${sourceName} has no criteria list.`);
}
const seen = new Set<string>();
let cruxCount = 0;
for (const [index, entry] of doc.criteria.entries()) {
const criterion = entry as Partial<RubricCriterion> | null;
if (typeof criterion !== 'object' || criterion === null) {
return fail(`${sourceName} criteria[${index}] is not a mapping.`);
}
if (typeof criterion.id !== 'string' || criterion.id.length === 0) {
return fail(`${sourceName} criteria[${index}] has no id.`);
}
if (seen.has(criterion.id)) {
return fail(`${sourceName} has a duplicate criterion id: ${criterion.id}.`);
}
seen.add(criterion.id);
if (typeof criterion.guideline !== 'string' || criterion.guideline.trim().length === 0) {
return fail(`${sourceName} criterion ${criterion.id} has no guideline.`);
}
if (typeof criterion.category !== 'string' || criterion.category.length === 0) {
return fail(`${sourceName} criterion ${criterion.id} has no category.`);
}
if (criterion.severity === 'crux') cruxCount += 1;
}
if (cruxCount > 2) {
return fail(`${sourceName} designates ${cruxCount} Crux criteria; the cap is two.`);
}
return doc as RubricsDoc;
}
/** Grader-facing view: guideline and elaboration only, severity-blind. */
function renderCriteriaMarkdown(doc: RubricsDoc): string {
const sections = doc.criteria.map((criterion) => {
const parts = [`### Criterion: ${criterion.id}`, criterion.guideline.trim()];
if (criterion.elaboration?.trim()) parts.push(criterion.elaboration.trim());
return parts.join('\n\n');
});
return sections.join('\n\n') + '\n';
}
function stage(taskDir: string): void {
const testsDir = join(taskDir, 'tests');
if (!existsSync(testsDir)) {
return fail(`${relative(TOOLKIT_ROOT, taskDir)} has no tests/ directory.`);
}
const rubricPath = resolveRubricPath(testsDir);
const sourceName = `tests/${basename(rubricPath)}`;
let parsed: unknown;
try {
parsed = parseYaml(readFileSync(rubricPath, 'utf8'));
} catch (error) {
return fail(`${sourceName} is not readable YAML: ${(error as Error).message}`);
}
const doc = toRubricsDoc(parsed, sourceName);
const created: string[] = [];
const writeStaged = (name: string, content: string, mode?: number): void => {
writeFileSync(join(testsDir, name), content, mode ? { mode } : undefined);
created.push(name);
};
writeStaged('rubric-criteria.md', renderCriteriaMarkdown(doc));
writeStaged(
'rubric-criteria.json',
JSON.stringify(
{
task: doc.task,
// severity feeds render-rubric-grade.py's severity-weighted
// aggregation; dimensions ride along for offline slicing. The grader
// never sees this file — severity-blindness lives in
// rubric-criteria.md.
criteria: doc.criteria.map((criterion) => ({
id: criterion.id,
category: criterion.category,
severity: criterion.severity ?? null,
dimensions: criterion.dimensions ?? [],
})),
},
null,
2
) + '\n'
);
// The renderer is a shared asset. Sync it so the regrade runs the current
// weights; scripts/harbor-regrade performs the same self-heal.
const rendererSource = join(SHARED_DIR, 'render-rubric-grade.py');
const rendererDest = join(testsDir, 'render-rubric-grade.py');
if (!existsSync(rendererSource)) {
return fail('task-shared/render-rubric-grade.py is missing from this toolkit.');
}
if (!existsSync(rendererDest) || sha256(rendererDest) !== sha256(rendererSource)) {
copyFileSync(rendererSource, rendererDest);
created.push('render-rubric-grade.py');
}
writeFileSync(join(testsDir, STAGE_MANIFEST), JSON.stringify({ created }, null, 2) + '\n');
if (!existsSync(join(testsDir, 'grader-context.md'))) {
log.warn(
'tests/grader-context.md is missing. The rubric grader modes read it beside the ' +
'criteria; write it before running a rubric-mode regrade or packaging the task.'
);
}
log.info(
{ source: sourceName, criteria: doc.criteria.length, staged: created },
'Staged the atomic-rubric grading copies.'
);
}
function restore(taskDir: string): void {
const testsDir = join(taskDir, 'tests');
const manifestPath = join(testsDir, STAGE_MANIFEST);
if (!existsSync(manifestPath)) {
log.warn('No staging manifest found; nothing to restore.');
return;
}
let names: string[] = [];
try {
const manifest = JSON.parse(readFileSync(manifestPath, 'utf8')) as { created?: unknown };
if (Array.isArray(manifest.created)) {
names = manifest.created.filter((n): n is string => typeof n === 'string');
}
} catch {
return fail(`${STAGE_MANIFEST} is unreadable; remove the staged files by hand.`);
}
for (const name of names) {
// Only ever files this script wrote into tests/ — refuse anything else.
if (name.includes('/') || name.includes('..')) continue;
const filePath = join(testsDir, name);
if (existsSync(filePath)) rmSync(filePath);
}
rmSync(manifestPath);
log.info({ removed: names }, 'Removed the staged grading copies.');
}
// The yaml package reaches containers created after it joined package.json;
// a container created earlier has every other dependency but not this one,
// so resolve it at run time and say what to do instead of crashing.
let parseYaml: (src: string) => unknown;
try {
const requireFromHere = createRequire(fileURLToPath(import.meta.url));
({ parse: parseYaml } = requireFromHere('yaml') as { parse: (src: string) => unknown });
} catch {
fail(
'The yaml package is not installed in this container. Rebuild the Authoring ' +
'container ("Dev Containers: Rebuild Container"), or run npm install in the toolkit root.'
);
}
const taskDir = resolveTaskDir(String(argv._[0]));
if (argv.restore) restore(taskDir);
else stage(taskDir);

View File

@@ -0,0 +1,146 @@
/**
* stamp-trial-inputs.ts — record, at trial LAUNCH time, the checksums of the
* task inputs a harbor run is about to execute against, and stamp them into
* the trial directories the run produces.
*
* Why launch time: copy-reference-run.ts used to capture checksums at COPY
* time, which misses the headline staleness ordering — run trials, edit the
* prompt, then copy the runs — and records the post-edit hashes (a genuinely
* stale run then reads `fresh`). Harbor creates trial dirs itself (and, on
* the daytona backend, populates them only at download after the trial), so
* the earliest host-side point to capture is the moment `scripts/harbor-run`
* launches: `capture` snapshots the inputs to a temp file before harbor
* starts, and `apply` copies that snapshot into each trial dir once the job
* directory exists. copy-reference-run.ts then prefers this run-time record
* over its own capture-at-copy fallback.
*
* Usage (normally invoked by scripts/harbor-run, not by hand):
* npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>
* npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]
*
* Ships in the worker toolkit (see raccoon-worker-toolkit/package-worker-toolkit.ts),
* so it must only import from its shipped file set — same constraint as
* copy-reference-run.ts.
*/
import './lib/check-devcontainer';
import { existsSync, readFileSync, readdirSync, statSync, writeFileSync } from 'fs';
import { fileURLToPath } from 'node:url';
import { basename, join, resolve } from 'path';
import {
INPUT_CHECKSUMS_FILENAME,
captureTaskInputs,
readTaskInputChecksums,
} from './lib/input-checksums';
function usage(): never {
console.error(
[
'Usage:',
' npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>',
' npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]',
].join('\n')
);
process.exit(1);
}
/** Snapshot the task inputs as they stand at launch; write the record to `outFile`. */
export function captureCommand(taskDir: string, outFile: string): void {
if (!existsSync(taskDir)) {
console.error(`Error: task dir ${taskDir} does not exist`);
process.exit(1);
}
// taskSlug scopes `apply` to this task's trial dirs (concurrent harbor-runs
// share harbor-jobs/) and makes any mis-routed stamp diagnosable later.
const record = { ...captureTaskInputs(taskDir, 'run'), taskSlug: basename(resolve(taskDir)) };
writeFileSync(outFile, JSON.stringify(record, null, 2) + '\n');
}
/**
* Does this trial dir belong to the task the capture was taken from? Two
* signals, strongest first:
*
* 1. result.json `task_name` — written per-trial by harbor with the FULL,
* unambiguous slug (org-prefixed for hub-published tasks). Authoritative
* when readable, exactly as copy-reference-run.ts resolves trials.
* 2. The trial DIRNAME's `<prefix>__<trialId>` prefix — harbor TRUNCATES
* long slugs here, so the test is "the prefix is a truncation of the
* slug", not equality. (Two tasks sharing a truncated prefix are told
* apart by signal 1; the dirname alone can't distinguish them.)
*/
function trialBelongsToTask(trialDir: string, entry: string, slug: string): boolean {
const sep = entry.lastIndexOf('__');
if (sep === -1) return false;
const resultPath = join(trialDir, 'result.json');
if (existsSync(resultPath)) {
try {
const taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown })
.task_name;
if (typeof taskName === 'string' && taskName.length > 0) {
return taskName.replace(/^[^/]+\//, '') === slug;
}
} catch {
// Unparseable result.json — fall through to the dirname prefix.
}
}
const prefix = entry.substring(0, sep);
return prefix === slug || slug.startsWith(prefix);
}
/**
* Copy a launch-time capture into the capture's OWN task's trial dirs under
* the given harbor job dir(s). Trial dirs are the `<slug>__<trialId>`
* subdirectories harbor creates; anything else (stray files, harbor's own
* metadata) is skipped — and so is any trial belonging to a DIFFERENT task:
* several harbor-run invocations can share a cwd, and stamping another
* task's trials with this capture's hashes would fabricate `capturedBy:
* 'run'` evidence for inputs that task never ran against. An existing record
* is left alone — it can only be from an earlier stamp of the same trial.
*/
export function applyCommand(captureFile: string, jobDirs: string[]): number {
const record = readTaskInputChecksums(captureFile);
if (!record || typeof record.taskSlug !== 'string' || record.taskSlug.length === 0) {
console.error(
`Error: ${captureFile} is not a readable input-checksums capture with a taskSlug`
);
process.exit(1);
}
const slug = record.taskSlug;
const raw = readFileSync(captureFile, 'utf-8');
let stamped = 0;
for (const jobDir of jobDirs) {
if (!existsSync(jobDir)) continue;
for (const entry of readdirSync(jobDir)) {
const trialDir = join(jobDir, entry);
if (!entry.includes('__') || !statSync(trialDir).isDirectory()) continue;
if (!trialBelongsToTask(trialDir, entry, slug)) continue;
const dest = join(trialDir, INPUT_CHECKSUMS_FILENAME);
if (existsSync(dest)) continue;
writeFileSync(dest, raw);
stamped++;
console.log(`Stamped ${dest}`);
}
}
return stamped;
}
// Main. Guarded so the test file can import the commands without running them.
if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1])) {
const [command, ...rest] = process.argv.slice(2);
if (command === 'capture') {
const outIdx = rest.indexOf('--out');
const taskDir = rest.filter((a, i) => i !== outIdx && i !== outIdx + 1)[0];
const outFile = outIdx !== -1 ? rest[outIdx + 1] : undefined;
if (!taskDir || !outFile) usage();
captureCommand(taskDir, outFile);
} else if (command === 'apply') {
const [captureFile, ...jobDirs] = rest;
if (!captureFile || jobDirs.length === 0) usage();
const stamped = applyCommand(captureFile, jobDirs);
console.log(`Stamped ${stamped} trial dir(s) with launch-time input checksums`);
} else {
usage();
}
}

View File

@@ -0,0 +1,93 @@
#!/usr/bin/env python3
"""str_replace_editor — CLI-as-MCP wrapper around the vendored EditTool.
This is the "CLI-as-MCP" delivery of the `str_replace_editor` tool: the agent
(which has ONLY the bash tool) invokes this script and passes the tool's
arguments as one JSON object on stdin. The actual editing logic is the vendored
`EditTool` under str_replace_editor_vendor/ (see VENDORED.md) — we add no
behavior, we only:
* instantiate it with run_command_preexec_fn=None (the class's own documented
way to skip its uid/gid-1000 demotion, which would break writes in our
sandbox where the workspace is owned by the agent user); and
* adapt structured stdin-JSON <-> a bash-invokable CLI.
stdin: one JSON object, e.g.
{"command":"view","path":"/workspace/app/models/x.rb"}
{"command":"view","path":"/workspace/x.rb","view_range":[1,40]}
{"command":"str_replace","path":"/workspace/x.rb","old_str":"a","new_str":"b"}
{"command":"create","path":"/workspace/new.rb","file_text":"..."}
{"command":"insert","path":"/workspace/x.rb","insert_line":10,"insert_text":"..."}
stdout: the tool's result text (exit 0). stderr + exit 1: a tool error message.
"""
import asyncio
import json
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from str_replace_editor_vendor.base import ToolError # noqa: E402
from str_replace_editor_vendor.edit import EditTool # noqa: E402
# The keyword-only params the vendored EditTool.__call__ accepts.
_ACCEPTED = {
"command", "path", "file_text", "view_range",
"old_str", "new_str", "insert_text", "insert_line",
}
async def _run(payload: dict):
# Reject unknown keys instead of silently dropping them: a typo like
# `old_string` (vs `old_str`) should be a clear argument error, not a
# confusing failure deeper inside EditTool with the param silently missing.
unknown = set(payload) - _ACCEPTED
if unknown:
raise ToolError(
f"unknown argument(s): {', '.join(sorted(unknown))}. "
f"accepted keys: {', '.join(sorted(_ACCEPTED))}."
)
kwargs = dict(payload)
if "command" not in kwargs or "path" not in kwargs:
raise ToolError("Both `command` and `path` are required.")
# run_command_preexec_fn=None → no uid/gid demotion (see module docstring).
tool = EditTool(run_command_preexec_fn=None)
return await tool(**kwargs)
def main() -> int:
raw = sys.stdin.read()
if not raw.strip():
sys.stderr.write("str_replace_editor: expected a JSON object on stdin\n")
return 2
try:
payload = json.loads(raw)
except json.JSONDecodeError as e:
sys.stderr.write(f"str_replace_editor: invalid JSON on stdin: {e}\n")
return 2
if not isinstance(payload, dict):
sys.stderr.write("str_replace_editor: stdin JSON must be an object\n")
return 2
try:
result = asyncio.run(_run(payload))
except ToolError as e:
sys.stderr.write((e.message or "tool error") + "\n")
return 1
except TypeError as e:
# e.g. an unexpected/duplicate kwarg shape — surface like a tool error.
sys.stderr.write(f"str_replace_editor: bad arguments: {e}\n")
return 1
# EditTool returns a (CLI)Result with .output / .error / .base64_image / .system
if getattr(result, "error", None):
sys.stderr.write(result.error if result.error.endswith("\n") else result.error + "\n")
if getattr(result, "system", None):
sys.stderr.write(f"[system] {result.system}\n")
out = getattr(result, "output", None) or ""
if getattr(result, "base64_image", None):
out += "\n(image content omitted in CLI mode)"
if out:
sys.stdout.write(out if out.endswith("\n") else out + "\n")
return 1 if getattr(result, "error", None) else 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -0,0 +1 @@
"""Vendored verbatim — do not edit. See VENDORED.md for provenance."""

View File

@@ -0,0 +1,49 @@
from dataclasses import dataclass, fields, replace
@dataclass(kw_only=True, frozen=True)
class ToolResult:
"""Represents the result of a tool execution."""
output: str | None = None
error: str | None = None
base64_image: str | None = None
system: str | None = None
def __bool__(self):
return any(getattr(self, field.name) for field in fields(self))
def __add__(self, other: "ToolResult"):
def combine_fields(field: str | None, other_field: str | None, concatenate: bool = True):
if field and other_field:
if concatenate:
return field + other_field
raise ValueError("Cannot combine tool results")
return field or other_field
return ToolResult(
output=combine_fields(self.output, other.output),
error=combine_fields(self.error, other.error),
base64_image=combine_fields(self.base64_image, other.base64_image, False),
system=combine_fields(self.system, other.system),
)
def replace(self, **kwargs):
"""Returns a new ToolResult with the given fields replaced."""
return replace(self, **kwargs)
# QUESTION(simon): What's our intent behind differentiating here?
class CLIResult(ToolResult):
"""A ToolResult that can be rendered as a CLI output."""
class ToolFailure(ToolResult):
"""A ToolResult that represents a failure."""
class ToolError(Exception):
"""Raised when a tool encounters an error."""
def __init__(self, message):
self.message = message

View File

@@ -0,0 +1,476 @@
import asyncio
import base64
import shlex
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, get_args
from .base import CLIResult, ToolError, ToolResult
from .run import demote, maybe_truncate, run
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
Command = Literal[
"view",
"create",
"str_replace",
"insert",
]
SNIPPET_LINES: int = 4
MAX_RESPONSE_LEN: int = 16000
class EditTool:
"""
An filesystem editor tool that allows the agent to view, create, and edit files.
The tool parameters are defined by Anthropic and are not editable.
"""
def __init__(self, run_command_preexec_fn=demote):
"""
Initialize the EditTool.
Args:
run_command_preexec_fn: Function to run in child process before executing
shell commands via the run() utility.
Defaults to demote() which drops privileges to uid/gid 1000.
Pass None to skip preexec, or any callable for custom behavior.
"""
self._run_command_preexec_fn = run_command_preexec_fn
async def __call__(
self,
*,
command: Command,
path: str,
file_text: str | None = None,
view_range: list[int] | None = None,
old_str: str | None = None,
new_str: str | None = None,
insert_text: str | None = None,
insert_line: int | None = None,
):
_path = Path(path)
self.validate_path(command, _path)
if command == "view":
return await self.view(_path, view_range)
elif command == "create":
if file_text is None:
raise ToolError("Parameter `file_text` is required for command: create")
await self.write_file(_path, file_text)
return ToolResult(output=f"File created successfully at: {_path}")
elif command == "str_replace":
if old_str is None:
raise ToolError("Parameter `old_str` is required for command: str_replace")
return await self.str_replace(_path, old_str, new_str)
elif command == "insert":
if insert_line is None:
raise ToolError("Parameter `insert_line` is required for command: insert")
if insert_text is None:
raise ToolError("Parameter `insert_text` is required for command: insert")
return await self.insert(_path, insert_line, insert_text)
raise ToolError(
f"Unrecognized command {command}. The allowed commands for the {self.name} tool are: {', '.join(get_args(Command))}"
)
def validate_path(self, command: str, path: Path):
"""
Check that the path/command combination is valid.
"""
# Check if its an absolute path
if not path.is_absolute():
suggested_path = Path("") / path
raise ToolError(
f"The path {path} is not an absolute path, it should start with `/`. Maybe you meant {suggested_path}?"
)
# Check if path exists
if not path.exists() and command != "create":
raise ToolError(f"The path {path} does not exist. Please provide a valid path.")
if path.exists() and command == "create":
raise ToolError(f"File already exists at: {path}. Cannot overwrite files using command `create`.")
# Check if the path points to a directory
if path.is_dir():
if command != "view":
raise ToolError(
f"The path {path} is a directory and only the `view` command can be used on directories"
)
async def view(self, path: Path, view_range: list[int] | None = None):
"""Implement the view command"""
if path.is_dir():
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to a directory.")
_, stdout, stderr = await run(
rf"find {path} -maxdepth 2 -not -path '*/\.*'", preexec_fn=self._run_command_preexec_fn
)
if not stderr:
stdout = f"Here's the files and directories up to 2 levels deep in {path}, excluding hidden items:\n{stdout}\n"
return CLIResult(output=stdout, error=stderr)
image_extensions = {'.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tiff', '.tif', '.webp', '.svg', '.ico'}
if path.suffix.lower() in image_extensions:
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to an image file.")
try:
image_bytes = path.read_bytes()
base64_encoded = base64.b64encode(image_bytes).decode()
return CLIResult(
output=f"Displaying image file: {path}",
base64_image=base64_encoded
)
except Exception as e:
raise ToolError(f"Failed to read image file {path}: {e}") from None
file_content = await self.read_file(path, truncate_after=None)
file_text_lines = file_content.splitlines(keepends=True)
n_lines_file = len(file_text_lines) + (1 if file_content.endswith(("\n", "\r\n", "\r")) else 0)
if view_range:
if len(view_range) != 2 or not all(isinstance(i, int) for i in view_range):
raise ToolError("Invalid `view_range`. It should be a list of two integers.")
init_line, final_line = view_range
if init_line < 1 or init_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its first element `{init_line}` should be within the range of lines of the file: {[1, n_lines_file]}"
)
if final_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be smaller than the number of lines in the file: `{n_lines_file}`"
)
if final_line != -1 and final_line < init_line:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be larger or equal than its first `{init_line}`"
)
# Extract only the requested lines
if final_line != -1:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) : view_range[1]]
else:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) :]
# Join without modifying the original line endings
file_content = "".join(selected_lines)
file_content = process_view_output_str(
file_text=file_content,
path=str(path),
total_path_lines=n_lines_file,
max_resp_ln=MAX_RESPONSE_LEN,
view_range=(view_range[0], view_range[1]) if view_range else None,
)
return CLIResult(output=file_content)
async def str_replace(self, path: Path, old_str: str, new_str: str | None):
"""Implement the str_replace command, which replaces old_str with new_str in the file content"""
# Read the file content
file_content = await self.read_file(path, truncate_after=None)
new_str = new_str if new_str is not None else ""
# Check if old_str is unique in the file
occurrences = file_content.count(old_str)
if occurrences == 0:
raise ToolError(f"No replacement was performed, old_str `{old_str}` did not appear verbatim in {path}.")
elif occurrences > 1:
file_content_lines = file_content.split("\n")
lines = [idx + 1 for idx, line in enumerate(file_content_lines) if old_str in line]
raise ToolError(
f"No replacement was performed. Multiple occurrences of old_str `{old_str}` in lines {lines}. Please ensure it is unique"
)
# Replace old_str with new_str
new_file_content = file_content.replace(old_str, new_str)
# Write the new content to the file
await self.write_file(path, new_file_content)
# Create a snippet of the edited section
replacement_line = file_content.split(old_str)[0].count("\n")
start_line = max(0, replacement_line - SNIPPET_LINES)
end_line = replacement_line + SNIPPET_LINES + new_str.count("\n")
snippet = "\n".join(new_file_content.split("\n")[start_line : end_line + 1])
# Prepare the success message
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(snippet, f"a snippet of {path}", start_line + 1)
success_msg += "Review the changes and make sure they are as expected. Edit the file again if necessary."
return CLIResult(output=success_msg)
async def insert(self, path: Path, insert_line: int, new_str: str):
"""Implement the insert command, which inserts new_str at the specified line in the file content."""
file_text = await self.read_file(path, truncate_after=None)
file_text_lines = file_text.split("\n")
n_lines_file = len(file_text_lines)
if insert_line < 0 or insert_line > n_lines_file:
raise ToolError(
f"Invalid `insert_line` parameter: {insert_line}. It should be within the range of lines of the file: {[0, n_lines_file]}"
)
new_str_lines = new_str.split("\n")
new_file_text_lines = file_text_lines[:insert_line] + new_str_lines + file_text_lines[insert_line:]
snippet_lines = (
file_text_lines[max(0, insert_line - SNIPPET_LINES) : insert_line]
+ new_str_lines
+ file_text_lines[insert_line : insert_line + SNIPPET_LINES]
)
new_file_text = "\n".join(new_file_text_lines)
snippet = "\n".join(snippet_lines)
await self.write_file(path, new_file_text)
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(
snippet,
"a snippet of the edited file",
max(1, insert_line - SNIPPET_LINES + 1),
)
success_msg += "Review the changes and make sure they are as expected (correct indentation, no duplicate lines, etc). Edit the file again if necessary."
return CLIResult(output=success_msg)
async def read_file(self, path: Path, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Read the content of a file from a given path; raise a ToolError if an error occurs."""
try:
code, out, err = await run(
f"cat {shlex.quote(str(path))}", truncate_after=truncate_after, preexec_fn=self._run_command_preexec_fn
)
if code != 0:
raise ToolError(f"Ran into {err} while trying to read {path}")
return out
except Exception as e:
print(e)
raise ToolError(f"Ran into {e} while trying to read {path}") from None
async def write_file(self, path: Path, file: str):
"""Write the content of a file to a given path; raise a ToolError if an error occurs."""
try:
# Write using stdin to avoid argument size limits
process = await asyncio.create_subprocess_shell(
f"cat > {shlex.quote(str(path))}",
stdin=asyncio.subprocess.PIPE,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=self._run_command_preexec_fn,
)
stdout, stderr = await asyncio.wait_for(
process.communicate(input=file.encode('utf-8')),
timeout=120.0
)
if process.returncode != 0:
raise ToolError(f"Ran into {stderr.decode()} while trying to write to {path}")
except asyncio.TimeoutError:
raise ToolError(f"Timed out while trying to write to {path}")
except Exception as e:
raise ToolError(f"Ran into {e} while trying to write to {path}") from None
def _make_output(
self,
file_content: str,
file_descriptor: str,
init_line: int = 1,
expand_tabs: bool = True,
):
"""Generate output for the CLI based on the content of a file."""
file_content = maybe_truncate(file_content)
if expand_tabs:
file_content = file_content.expandtabs()
file_content = "\n".join([f"{i + init_line:6}\t{line}" for i, line in enumerate(file_content.split("\n"))])
return f"Here's the result of running `cat -n` on {file_descriptor}:\n" + file_content + "\n"
### AUX utilities
def add_line_numbers(text: str, includes_final_line: bool, n_first_line: int = 1) -> str:
"""
Given a string, returns the string with line numbers prepended to each line.
This function:
- Preserves the original line endings (CR, LF, or CRLF) of each line
- Adds a tab-separated line number prefix to each line
- If the text ends with any newline character (\n, \r\n, or \r), adds an
additional empty numbered line to represent the terminal empty line
"""
lines_with_endings = text.splitlines(keepends=True)
result = [f"{ind + n_first_line:6}\t{line_with_ending}" for ind, line_with_ending in enumerate(lines_with_endings)]
# Add an extra empty line with line number if original text ends with newline
if includes_final_line and text.endswith(("\n", "\r\n", "\r")):
result.append(f"{len(lines_with_endings) + n_first_line:6}\t")
return "".join(result)
def process_view_output_str(
file_text: str,
path: str,
total_path_lines: int,
max_resp_ln: int,
view_range: tuple[int, int] | None = None,
) -> str:
# Get header
header = f"Here's the content of {path} with line numbers"
if total_path_lines is not None and view_range is not None:
header += f" (which has a total of {total_path_lines} lines) with view_range={list(view_range)}"
# See if final line is included in the view_range
if view_range is None or view_range[1] == -1 or view_range[1] == total_path_lines:
includes_final_line = True
else:
includes_final_line = False
n_first_line = view_range[0] if view_range is not None else 1
# Truncate if needed
maybe_truncated_str = truncate_from_middle_v2(ss=file_text, max_len=max_resp_ln, n_line_offset=n_first_line - 1)
if isinstance(maybe_truncated_str, str):
# No truncation
file_text_with_line_numbers = add_line_numbers(
file_text,
includes_final_line=includes_final_line,
n_first_line=n_first_line,
)
else:
# Truncation occurred
before_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.before_lines)),
includes_final_line=False,
n_first_line=n_first_line,
)
if maybe_truncated_str.single_line:
file_text_with_line_numbers = before_with_line_numbers
else:
after_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.after_lines)),
includes_final_line=includes_final_line,
n_first_line=1 + maybe_truncated_str.truncated_end_line,
)
file_text_with_line_numbers = (
before_with_line_numbers + f"\t{maybe_truncated_str.truncation_msg}" + after_with_line_numbers
)
# Add context-aware truncation message
if view_range is not None:
# User already using view_range, suggest adjusting it
truncation_note = "\n<response clipped><NOTE>To save on context only part of the view range has been shown. You can adjust the view_range parameters or use `grep -n` to find specific content.</NOTE>"
else:
# User viewing whole file, suggest view_range or grep
truncation_note = "\n<response clipped><NOTE>To save on context only part of this file has been shown to you. You can use view_range=[start_line, end_line] to see specific sections, or use `grep -n` to find what you're looking for.</NOTE>"
file_text_with_line_numbers += truncation_note
return f"{header}:\n{file_text_with_line_numbers}"
@dataclass
class TruncatedString:
# Blocks
before_lines: list[str]
middle_lines: list[str]
after_lines: list[str]
# Line numbers (starting from 1)
truncated_start_line: int
truncated_end_line: int
# Truncation msg
truncation_msg: str
single_line: bool
def as_str(self, lines: list[str]) -> str:
return "".join(lines)
@property
def full_truncated_str(self) -> str:
return "".join(self.before_lines + [self.truncation_msg] + self.after_lines)
def truncate_from_middle_v2(ss: str, max_len: int, n_line_offset: int = 0) -> "str | TruncatedString":
"""
If no truncation is needed, returns the original string.
If truncation is needed, returns TruncatedString
"""
# No truncation needed
if len(ss) <= max_len:
return ss
# Single line
lines_with_endings = ss.splitlines(True)
if len(lines_with_endings) == 1:
chars_per_side = max(1, max_len // 2)
truncated_char_count = len(ss) - (chars_per_side * 2)
truncation_msg = f"...< truncated {truncated_char_count} characters >..."
before_lines = [ss[:chars_per_side] + truncation_msg + ss[-chars_per_side:]]
return TruncatedString(
before_lines=before_lines,
middle_lines=[],
after_lines=[],
truncated_start_line=1 + n_line_offset,
truncated_end_line=1 + n_line_offset,
truncation_msg=truncation_msg,
single_line=True,
)
# Line truncation
current_len = 0
before_lines = []
middle_lines = deque(lines_with_endings)
after_lines = deque([])
while current_len < max_len and len(middle_lines) > 1:
# Before
before_candidate_line = middle_lines[0]
if len(before_candidate_line) + current_len <= max_len:
before_lines.append(middle_lines.popleft())
current_len += len(before_candidate_line)
else:
break
# After
if len(middle_lines) > 1:
after_candidate_line = middle_lines[-1]
if len(after_candidate_line) + current_len <= max_len:
after_lines.appendleft(middle_lines.pop())
current_len += len(after_candidate_line)
else:
break
# Find truncated lines
first_truncated_line = 1 + len(before_lines) + n_line_offset
last_truncated_line = first_truncated_line + len(middle_lines) - 1
if ss.endswith(("\n", "\r", "\r\n")) and len(after_lines) == 0:
last_truncated_line += 1
# Create truncation msg
if first_truncated_line == last_truncated_line:
truncation_msg = f"< truncated line {first_truncated_line} >"
else:
truncation_msg = f"< truncated lines {first_truncated_line}-{last_truncated_line} >"
if len(after_lines) != 0:
if before_lines[0].endswith("\r\n"):
truncation_msg += "\r\n"
elif before_lines[0].endswith("\r"):
truncation_msg += "\r"
else:
truncation_msg += "\n"
return TruncatedString(
# Blocks
before_lines=before_lines,
middle_lines=list(middle_lines),
after_lines=list(after_lines),
# Line numbers (starting from 1)
truncated_start_line=first_truncated_line,
truncated_end_line=last_truncated_line,
# Truncation msg
truncation_msg=truncation_msg,
single_line=False,
)

View File

@@ -0,0 +1,66 @@
"""Utility to run shell commands asynchronously with a timeout."""
import asyncio # noqa -- swapping to trio would be beneficial, but not blocking atm
import os
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
MAX_RESPONSE_LEN: int = 16000
def maybe_truncate(content: str, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Truncate content and append a notice if content exceeds the specified length."""
return (
content
if not truncate_after or len(content) <= truncate_after
else content[:truncate_after] + TRUNCATED_MESSAGE
)
def demote():
"""Drop privileges to uid/gid 1000 for security.
This function is intended to be used as a preexec_fn in subprocess calls
to ensure commands run with reduced privileges.
"""
os.setgid(1000)
os.setuid(1000)
async def run(
cmd: str,
timeout: float | None = 120.0, # seconds # noqa: ASYNC109
truncate_after: int | None = MAX_RESPONSE_LEN,
preexec_fn=demote,
):
"""Run a shell command asynchronously with a timeout.
Args:
cmd: Command to execute
timeout: Command timeout in seconds
truncate_after: Maximum response length before truncation
preexec_fn: Function to run in child process before exec (default: demote).
Pass None to skip preexec, or any callable for custom behavior.
Returns:
Tuple of (return_code, stdout, stderr)
"""
process = await asyncio.create_subprocess_shell(
cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=preexec_fn,
)
try:
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout)
return (
process.returncode or 0,
maybe_truncate(stdout.decode(), truncate_after=truncate_after),
maybe_truncate(stderr.decode(), truncate_after=truncate_after),
)
except TimeoutError as exc:
try:
process.kill()
except ProcessLookupError:
pass
raise TimeoutError(f"Command '{cmd}' timed out after {timeout} seconds") from exc

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,20 @@
# Your actual toolset (this overrides any earlier tool guidance above)
This harness gives you exactly two ways to act, both through the `Bash` tool:
1. **Shell commands** for everything read-only and for running things: view and search files with `cat`, `sed -n`, `grep -rn`, `find`, `ls`; run tests; run `git`; etc.
2. **A `str_replace_editor` file editor**, which you invoke from Bash by piping ONE JSON object on stdin to `/opt/agent-cli/str_replace_editor`. Use a quoted heredoc so backslashes and quotes survive:
`/opt/agent-cli/str_replace_editor <<'EDITOR'` then a line of JSON then `EDITOR`
The JSON `"command"` field selects the operation:
- `view` — view a file (optionally `"view_range":[start,end]`) or list a directory: `{"command":"view","path":"/abs/file.rb"}`
- `create` — create a NEW file (fails if it exists): `{"command":"create","path":"/abs/new.rb","file_text":"..."}`
- `str_replace` — replace a UNIQUE substring: `{"command":"str_replace","path":"/abs/file.rb","old_str":"...","new_str":"..."}`
- `insert` — insert text after a line: `{"command":"insert","path":"/abs/file.rb","insert_line":N,"insert_text":"..."}`
Paths must be absolute. Inside JSON strings, escape newlines as `\n` and double-quotes as `\"`.
There are **no** `Read`, `Grep`, `Glob`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite`, or `AskUserQuestion` tools — `Bash` is your only built-in tool. So disregard the earlier "Prefer the dedicated file/search tools over shell commands" guidance and the Memory section's "use the Write tool" instruction: those tools are not available in this harness. Search and read with shell commands; view, create, and edit files with `str_replace_editor`.
There is also no tool for asking the user an interactive question. If you need to ask the user something, or raise a concern about the request before acting on it, put it in your normal text response.

View File

@@ -0,0 +1,4 @@
## Browser
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
`require("playwright")` resolvable (CommonJS — `import` will not find it).

View File

@@ -0,0 +1,7 @@
## Correction to the toolset above: you also have `Read`
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
edit files with `str_replace_editor`.

View File

@@ -0,0 +1,88 @@
#!/usr/bin/env python3
"""validate_task_dir.py — say WHY harbor will not accept a task directory.
``harbor run -p <dir>`` silently reinterprets a directory that fails task
validation as a *dataset* of tasks, finds none inside, and dies with
``ValueError: Either datasets or tasks must be provided.`` — a message naming
neither the path nor the missing file. ``scripts/harbor-run`` calls this first so
the author reads "tests/test.sh is missing" instead.
Must run under HARBOR'S interpreter (its uv-tool venv), not any python3.11+: it
imports harbor to reuse ``Task.is_valid_dir``, the exact predicate the CLI
branches on, so the two cannot drift.
Usage: validate_task_dir.py <task-dir> [--disable-verification]
Prints ``verdict=valid`` or ``verdict=invalid`` on stdout; the reason goes to
stderr. Callers must gate on the stdout verdict, never on the exit code alone —
an interpreter that cannot run this file at all also exits non-zero.
Exit 0 = valid, 1 = invalid, 2 = the check could not run.
"""
import sys
from pathlib import Path
def reason(task_dir: Path, disable_verification: bool) -> str | None:
"""Return why harbor rejects task_dir, or None if it accepts it."""
from harbor.models.task.config import TaskConfig
from harbor.models.task.paths import TaskPaths
from harbor.models.task.task import Task
if Task.is_valid_dir(task_dir, disable_verification=disable_verification):
return None
paths = TaskPaths(task_dir)
if not paths.config_path.exists():
return f"{paths.config_path} is missing."
if not paths.environment_dir.exists():
return f"{paths.environment_dir} is missing."
try:
config = TaskConfig.model_validate_toml(paths.config_path.read_text())
except Exception as exc:
return f"{paths.config_path} does not parse as a task config: {exc}"
# A stepped task carries no root instruction.md, so only the shape harbor
# checks may be asserted here — hence steps first, root instruction last.
if disable_verification:
for step in config.steps or []:
if not paths.step_dir(step.name).exists():
return f"{paths.step_dir(step.name)} is missing."
if not paths.step_instruction_path(step.name).exists():
return f"{paths.step_instruction_path(step.name)} is missing."
if not config.steps and not paths.instruction_path.exists():
return f"{paths.instruction_path} is missing."
else:
# Private, but it owns the instruction/test diagnostics is_valid_dir discards.
try:
Task._validate_tests(config, paths)
except FileNotFoundError as exc:
return str(exc)
except AttributeError:
pass
return f"{task_dir} is not a task directory harbor recognizes."
def main() -> int:
args = sys.argv[1:]
disable_verification = "--disable-verification" in args
positional = [a for a in args if not a.startswith("-")]
if len(positional) != 1:
print(
f"usage: {sys.argv[0]} <task-dir> [--disable-verification]", file=sys.stderr
)
return 2
try:
why = reason(Path(positional[0]), disable_verification)
except Exception as exc:
print(f"validate_task_dir: check did not run ({exc})", file=sys.stderr)
return 2
if why is None:
print("verdict=valid")
return 0
print("verdict=invalid")
print(why, file=sys.stderr)
return 1
if __name__ == "__main__":
sys.exit(main())

View File

@@ -0,0 +1,41 @@
#!/bin/bash
# Welcome banner for raccoon dev containers
CYAN='\033[1;36m'
YELLOW='\033[1;33m'
GRAY='\033[0;90m'
RESET='\033[0m'
CONTAINER_TYPE="${1:-explore}"
if [ "$CONTAINER_TYPE" = "explore" ]; then
COLOR="$CYAN"
else
COLOR="$YELLOW"
fi
cat << 'RACCOON'
.----------------. .----------------. .----------------. .----------------. .----------------. .----------------. .-----------------.
| .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. |
| | _______ | || | __ | || | ______ | || | ______ | || | ____ | || | ____ | || | ____ _____ | |
| | |_ __ \ | || | / \ | || | .' ___ | | || | .' ___ | | || | .' `. | || | .' `. | || ||_ \|_ _| | |
| | | |__) | | || | / /\ \ | || | / .' \_| | || | / .' \_| | || | / .--. \ | || | / .--. \ | || | | \ | | | |
| | | __ / | || | / ____ \ | || | | | | || | | | | || | | | | | | || | | | | | | || | | |\ \| | | |
| | _| | \ \_ | || | _/ / \ \_ | || | \ `.___.'\ | || | \ `.___.'\ | || | \ `--' / | || | \ `--' / | || | _| |_\ |_ | |
| | |____| |___| | || ||____| |____|| || | `._____.' | || | `._____.' | || | `.____.' | || | `.____.' | || ||_____|\____| | |
| | | || | | || | | || | | || | | || | | || | | |
| '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' |
'----------------' '----------------' '----------------' '----------------' '----------------' '----------------' '----------------'
__ .-.
.-"` .`'. /\\|
_(\-/)_" , . ,\ /\\\/
{(#b^d#)} . ./, |/\\\/
`-.(Y).-` , | , |\.-`
/~/,_/~~~\,__.-`
////~ // ~\\
==`==` ==` ==`
------------------------------------------------
RACCOON