Project-2 baseline

This commit is contained in:
2026-10-04 21:19:23 -04:00
parent 0f04889edf
commit 213eb3c403
861 changed files with 1710 additions and 3363322 deletions

View File

@@ -159,6 +159,31 @@ if [ ! -f "$TASK_DIR/tests/test-commands.sh" ]; then
fi
fi
# 3. Companion service. strongsuit-app throws rather than degrades when the strongsuit_phx
# backend on :4201 is absent, so stage it for Dockerfile.strongsuit-app to COPY.
#
# Tracked, not gitignored: harbor-regrade and the benchmark sweep build task images without
# running this script, and a missing COPY source fails the build (as for browser-optin).
if [ "$MEMBER_LC" = "strongsuit-app" ]; then
PHX_SRC="$TOOLKIT_ROOT/repos/strongsuit_phx"
PHX_SEED_SRC="$TOOLKIT_ROOT/explore/scripts/seed"
PHX_STAGE="$TASK_DIR/environment/phx"
PHX_SEED_STAGE="$TASK_DIR/environment/phx-seed"
if [ -d "$PHX_SRC/.git" ] && [ -d "$PHX_SEED_SRC" ]; then
PHX_PIN=$(node -e "const r=require('$TOOLKIT_ROOT/toolkit.json').repos.find(x=>x.repo==='strongsuit_phx');process.stdout.write(r&&r.defaultCommit||'HEAD')" 2>/dev/null || echo HEAD)
PHX_SHA=$(resolve_pin "$PHX_SRC" "$PHX_PIN" "$TOOLKIT_ROOT/task-shared/commit-maps" strongsuit_phx) || exit 1
rm -rf "$PHX_STAGE" "$PHX_SEED_STAGE"
mkdir -p "$PHX_STAGE" "$PHX_SEED_STAGE"
git -C "$PHX_SRC" archive "$PHX_SHA" | tar -x --no-same-owner -C "$PHX_STAGE"
cp "$PHX_SEED_SRC"/*.sh "$PHX_SEED_STAGE"/
chmod +x "$PHX_SEED_STAGE"/*.sh
echo " Staged Phoenix backend: environment/phx/ (${PHX_SHA:0:9}) + environment/phx-seed/"
else
echo " WARN: strongsuit_phx sources not found under repos/ — the task image will have no" >&2
echo " backend on :4201, and the strongsuit-app pages that read it will throw." >&2
fi
fi
# Clean and recreate
rm -rf "$WORKSPACE"
mkdir -p "$WORKSPACE"

View File

@@ -30,11 +30,11 @@ from pathlib import Path
import atif_session
import browser_note
try:
from dnsjail import apply_dns_jail
from dnsjail import jailed
except ImportError: # no helper shipped -> no jail, rather than no trials
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
return None
def jailed(run): # type: ignore[misc]
return run
from harbor.agents.installed.codex import Codex
@@ -117,10 +117,10 @@ class SystemNodeCodex(Codex):
_has_browser = False
# Non-snapshot codex tasks run this class directly; the native-snapshot resume path
# overrides run() and applies the jail itself.
# overrides run() and carries its own @jailed.
@jailed
async def run(self, instruction, environment, context): # type: ignore[override]
self._refuse_shell_hostile_key()
await apply_dns_jail(self, environment)
await super().run(instruction, environment, context)
def _auth_json_setup(self, remote_auth_path: str) -> tuple[dict[str, str], str]:
@@ -343,11 +343,10 @@ class InlineSnapshotCodex(SystemNodeCodex):
# A real recorded codex session_meta line (with codex's base_instructions) is the
# most reliable seed for `resume`. The fixture lives in-repo (mounted into the
# devcontainer where this agent code runs); a captured host copy is a secondary
# source, and a synthesized minimal record is the final fallback.
# devcontainer where this agent code runs); a synthesized minimal record is the
# fallback.
_ROLLOUT_TEMPLATE_CANDIDATES = (
os.path.join(os.path.dirname(os.path.abspath(__file__)), "codex-rollout-template.jsonl"),
"/Users/nickheiner/.claude/jobs/e8fade29/tmp/codex-rollout-template.jsonl",
)
@@ -463,6 +462,7 @@ class NativeSnapshotCodex(SystemNodeCodex):
self.logger.warning("NativeSnapshotCodex: could not read session: %s", exc)
return ""
@jailed
async def run(self, instruction, environment, context): # type: ignore[override]
# NOTE: codex unconditionally declares its `tool_search` (MCP apps tool-
# discovery) tool, which the OpenAI API REJECTS for nano models with HTTP
@@ -471,7 +471,6 @@ class NativeSnapshotCodex(SystemNodeCodex):
# disabled_tools) suppress it as of codex 0.135, so nano models are NOT
# runnable under this harness. Use a mini (e.g. gpt-5.4-mini) for the small
# end instead. Non-nano models are unaffected.
await apply_dns_jail(self, environment)
session_text = await self._read_staged_session(environment)
if not session_text:
self.logger.info(

View File

@@ -489,6 +489,8 @@ function copyTrial(
}
rmSync(old, { recursive: true });
console.log(`Superseded ${supersede} (removed)`);
console.log(' Detector reports written before now may name the removed run. After your');
console.log(' last regrade, run submit-task.ts: it names each report to re-run.');
}
}

View File

@@ -4,14 +4,19 @@ Opt-in with RACCOON_DNS_JAIL=1. Runs from the agent's own turn rather than from
overlay — the allowlist comes from the proxy URL this process already holds (plus any hosts
RACCOON_DNS_JAIL_ALLOW adds), so nothing has to be injected into the container, and the jail works on every harbor backend. Deliberately
after agent-setup: a harness that downloads its CLI there still reaches the network to do it.
Covers the agent's turn and nothing else -- `jailed` lifts it again before harbor's verifier
phase, which shares the container and runs test suites we do not control.
"""
import functools
import logging
import os
import shlex
from typing import Any
JAIL = "/usr/local/bin/raccoon-dns-jail"
STATE = "/tmp/.dnsjail" # the container script's own state dir
_NO_SCRIPT = "raccoon-dns-jail: not in this image"
_URL_VARS = (
@@ -96,3 +101,44 @@ async def apply_dns_jail(agent: Any, environment: Any) -> None:
_log.warning(
"DNS jail: this task's image ships no resolver — the trial keeps normal network access"
)
async def lift_dns_jail(agent: Any, environment: Any) -> None:
"""Restore the container's own resolver once the agent's turn is over.
Harbor grades a timed-out or crashed agent turn too, so a jail left standing would
reach the verifier and change what the task's own test suite can do.
"""
if not dns_jail_enabled():
return
try:
# Keyed on the file the apply wrote, not on the baked script: a task image frozen
# before this feature has nothing to invoke, and must still end up unjailed.
await agent.exec_as_root(
environment,
command=(
f"if [ -s {STATE}/resolv.orig ]; then "
f"cat {STATE}/resolv.orig > /etc/resolv.conf; fi"
),
)
except Exception as exc:
# Raising here would replace whatever ended the agent's turn with this.
_log.warning("DNS jail: could not lift before the verifier (%s)", exc)
def jailed(run: Any) -> Any:
"""Wrap an agent `run()` so the jail covers exactly the agent's turn.
A decorator rather than two calls in the body: the pair is what matters, and an
`apply` whose `lift` was forgotten looks like a working trial.
"""
@functools.wraps(run)
async def wrapper(self, instruction, environment, context):
await apply_dns_jail(self, environment)
try:
return await run(self, instruction, environment, context)
finally:
await lift_dns_jail(self, environment)
return wrapper

View File

@@ -194,6 +194,9 @@ if [ ${#REF_DIRS[@]} -gt 1 ]; then
# each one is a sandbox that bills until something else reaps it.
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 143' TERM
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 130' INT
# One stamp for the whole fan-out: a second round is a NEW job to harbor, and it
# refuses (or silently resumes) a job dir whose name the previous round took.
ROUND_STAMP="$(date +%s)-$$"
IDX=0
TOTAL=${#REF_DIRS[@]}
while [ "$IDX" -lt "$TOTAL" ]; do
@@ -205,7 +208,7 @@ if [ ${#REF_DIRS[@]} -gt 1 ]; then
RUN_ID=$(basename "$REF")
# --job-name, not a nested output dir: children keep harbor's own
# harbor-jobs/<job>/<trial> shape. Indexed so duplicate args cannot collide.
JOB_NAME="regrade-$((IDX + 1))-$RUN_ID"
JOB_NAME="regrade-$ROUND_STAMP-$((IDX + 1))-$RUN_ID"
LOG="$OUT_BASE/$JOB_NAME.log"
echo " starting $RUN_ID (log: $LOG)"
"$SCRIPT_DIR/harbor-regrade" "$TASK_DIR_ABS" "$REF" \
@@ -246,6 +249,16 @@ if [ ${#REF_DIRS[@]} -gt 1 ]; then
echo "The rest are still in their job dirs; their logs above say why." >&2
fi
;;
*)
# A replaced holistic grade re-mints the run folder under its new reward and
# trial id, so every detector report keyed on run names is now out of date.
if [ -n "$REPLACE" ]; then
echo "" >&2
echo "Each replaced run's folder was re-minted under its new reward and trial id, so" >&2
echo "detector reports written before now may name runs that are gone. Run" >&2
echo "submit-task.ts: it names each report to re-run." >&2
fi
;;
esac
exit 0
fi
@@ -561,6 +574,16 @@ if [ ${#TRIALS[@]} -eq 0 ]; then
echo "Note: no trial directory under $JOB_DIR, so there is no grade." >&2
exit "$HARBOR_EXIT"
fi
# harbor exits 0 when a trial dies (e.g. a failed image build), so the reward is the signal.
UNGRADED=0
for _t in "${TRIALS[@]}"; do
if [ ! -f "$_t/verifier/reward.txt" ]; then
UNGRADED=$((UNGRADED + 1))
echo "Error: $(rel "$_t") produced no grade (no verifier/reward.txt)." >&2
[ -f "$_t/exception.txt" ] && echo " See $(rel "$_t")/exception.txt" >&2
fi
done
[ "$UNGRADED" -gt 0 ] && exit 1
if [ ${#TRIALS[@]} -gt 1 ]; then
echo "Note: ${#TRIALS[@]} trials under $JOB_DIR (-k grades one run repeatedly)." >&2
echo " Pick the one to keep and file it: $FILE_CMD <trial-dir>" >&2

View File

@@ -209,10 +209,11 @@ fi
# rebuild inputs (the workspace/ dir is gitignored) and every downstream
# consumer rebuilds from them, so anything uncaptured silently vanishes after
# packaging.
# The helper sits next to this script in a packed toolkit and under the
# toolkit's static scripts in the internal repo layout.
# The helper sits next to this script in a packed toolkit, and under toolkit/static/scripts
# in a repo checkout that consumes the toolkit as a submodule.
for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \
"$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do
"$REPO_ROOT/toolkit/static/scripts/check-workspace-sync.sh" \
"$REPO_ROOT/static/scripts/check-workspace-sync.sh"; do
if [ -f "$WS_SYNC" ]; then
bash "$WS_SYNC" "$TASK_DIR" || true
break
@@ -346,7 +347,7 @@ if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
exit 1
fi
# Task-image Claude Code floor. A task image installs Claude Code when it is first built and
# Task-image CLI floors. A task image installs Claude Code and Codex when it is first built and
# Docker reuses that layer on every later build, --force-build included, so an image built
# before the grader model's minimum CLI shipped fails every grade with "does not support this
# model" and the trial ends in RewardFileNotFoundError. Before a local docker trial, check the
@@ -354,6 +355,7 @@ fi
# cache, so harbor's build below installs a current CLI. Advisory: no docker, no daemon, no
# images, or RACCOON_SKIP_IMAGE_PREFLIGHT=1 means nothing happens. The floor follows
# GRADER_CLI_MIN in the shared test.sh; RACCOON_CLAUDE_CODE_MIN overrides it.
# Codex requires stable 0.158.0, matching task-shared/codex-grader.py.
CLAUDE_CODE_MIN="${RACCOON_CLAUDE_CODE_MIN:-}"
if [ -z "$CLAUDE_CODE_MIN" ]; then
for _ts in "$REPO_ROOT/task-shared/test.sh" "$REPO_ROOT/harbor-tasks/raccoon-shared/test.sh"; do
@@ -367,12 +369,22 @@ if [ "$ENV_TYPE" = "docker" ] && [ "${RACCOON_SKIP_IMAGE_PREFLIGHT:-0}" != "1" ]
&& command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
STALE_IMAGES=""
for img in $(docker images --format '{{.Repository}}:{{.Tag}}' 2>/dev/null | grep '^hb__' || true); do
v="$(docker run --rm --entrypoint claude "$img" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
[ -n "$v" ] || continue
if [ "$(printf '%s\n%s\n' "$CLAUDE_CODE_MIN" "$v" | sort -V | head -1)" != "$CLAUDE_CODE_MIN" ]; then
STALE_IMAGES="$STALE_IMAGES $img"
echo "Task image $img carries Claude Code $v; the grader needs $CLAUDE_CODE_MIN or newer. Removing it so the next build installs a current one." >&2
fi
for cli in claude codex; do
minimum="$CLAUDE_CODE_MIN"
[ "$cli" = codex ] && minimum=0.158.0
v="$(docker run --rm --entrypoint "$cli" "$img" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+[^[:space:]]*' | head -1 || true)"
# Keep Claude's existing handling of missing versions and suffixes.
if [ "$cli" = claude ]; then
v="$(printf '%s' "$v" | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' || true)"
[ -n "$v" ] || continue
fi
if ! [[ "$v" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] \
|| [ "$(printf '%s\n%s\n' "$minimum" "$v" | sort -V | head -1)" != "$minimum" ]; then
STALE_IMAGES="$STALE_IMAGES $img"
echo "Task image $img carries $cli ${v:-missing}; the grader needs stable $minimum or newer. Removing it so the next build installs a current one." >&2
break
fi
done
done
if [ -n "$STALE_IMAGES" ]; then
for img in $STALE_IMAGES; do
@@ -382,7 +394,7 @@ if [ "$ENV_TYPE" = "docker" ] && [ "${RACCOON_SKIP_IMAGE_PREFLIGHT:-0}" != "1" ]
done
# The stale install layer also lives in the build cache, where a rebuild would find it.
docker builder prune -af >/dev/null 2>&1 || true
echo "The task image rebuilds once with a current Claude Code; later runs reuse it." >&2
echo "The task image rebuilds once with current agent CLIs; later runs reuse it." >&2
fi
fi

View File

@@ -86,3 +86,14 @@ never a numeric magnitude, never points, never a cap or pinned score: the
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
Never describe how criteria combine into an overall score.>
`;
/** The title names the task, not its author: a leading `<EUID>-` (the worker's 12-char id,
* which workers put on their task dir) is dropped. Anything else is used as given. */
export function rubricTitleSlug(slug: string): string {
return slug.replace(/^(?=[A-Z0-9]*\d)[A-Z0-9]{12}-(?=\S)/, '');
}
/** The scaffold with `<task-slug>` filled in for a task whose name is already known. */
export function holisticRubricScaffoldFor(slug: string): string {
return HOLISTIC_RUBRIC_SCAFFOLD.replace('<task-slug>', rubricTitleSlug(slug));
}

View File

@@ -76,7 +76,10 @@ dnsjail_apply() {
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
# -u root: dnsmasq 2.80 (buster and older bases) drops to "nobody" and calls capset to
# retain CAP_NET_ADMIN, which docker's default cap set does not grant -- so it exits and
# the jail fails open on every such image.
dnsmasq -u root --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi

View File

@@ -53,6 +53,38 @@ _harness_trim() {
printf '%s' "${out:-$1}"
}
# Every env var a harness authenticates from, registry-derived so a new harness row is
# covered without touching this. ANTHROPIC_* unconditionally: it is what .env carries and
# what harbor-run hands the trial sandbox, registry or not.
_harness_credential_vars() {
local id key_env base_url_env proxy_path
printf '%s\n' ANTHROPIC_API_KEY ANTHROPIC_BASE_URL
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
if [ -n "$key_env" ]; then printf '%s\n' "$key_env"; fi
if [ -n "$base_url_env" ]; then printf '%s\n' "$base_url_env"; fi
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}
# Source .env into the CALLER's environment and trim what a harness reads its key from.
# For codex the live value is now the env var, not the auth file harness_write_auth
# cleans, so a raw `set -a; . .env` is the 401 all over again on a Windows-saved file.
harness_load_env() {
local file="${1:-${RACCOON_ENV_FILE:-/workspace/.env}}" v
if [ -f "$file" ]; then
set -a
# shellcheck disable=SC1090
. "$file" 2>/dev/null || true
set +a
fi
# Trimming twice is a no-op, so a var named by several rows needs no dedupe.
while read -r v; do
[ -n "$v" ] || continue
if [ -n "${!v:-}" ]; then
export "$v=$(_harness_trim "${!v}")"
fi
done < <(_harness_credential_vars)
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url

View File

@@ -0,0 +1,68 @@
import { existsSync, readFileSync, statSync } from 'fs';
const MB = 1024 * 1024;
/** At or above this the package is refused: GitHub warns at 50 MB and rejects at 100 MB. */
export const PATCH_MAX_BYTES = 49 * MB;
export const PATCH_WARN_BYTES = 10 * MB;
export const PATCH_BINARY_FILE_WARN_BYTES = 1 * MB;
export interface PatchSizeReport {
bytes: number;
errors: string[];
warnings: string[];
}
function fmt(bytes: number): string {
return `${(bytes / MB).toFixed(1)} MB`;
}
/** Binary files in a `git diff --binary` patch whose size exceeds the threshold. */
export function largeBinaryFiles(
patchText: string,
thresholdBytes = PATCH_BINARY_FILE_WARN_BYTES
): { path: string; bytes: number }[] {
const out: { path: string; bytes: number }[] = [];
const lines = patchText.split('\n');
let current = '';
for (let i = 0; i < lines.length; i++) {
const line = lines[i];
const header = /^diff --git a\/.* b\/(.*)$/.exec(line);
if (header) {
current = header[1];
continue;
}
if (line === 'GIT binary patch') {
const m = /^(?:literal|delta) (\d+)$/.exec(lines[i + 1] ?? '');
const bytes = m ? parseInt(m[1], 10) : 0;
if (bytes > thresholdBytes) out.push({ path: current, bytes });
}
}
return out;
}
export function checkWorkspacePatchSize(patchPath: string): PatchSizeReport {
const report: PatchSizeReport = { bytes: 0, errors: [], warnings: [] };
if (!existsSync(patchPath)) return report;
report.bytes = statSync(patchPath).size;
if (report.bytes >= PATCH_MAX_BYTES) {
report.errors.push(
`environment/workspace.patch is ${fmt(report.bytes)} — the limit is ${fmt(PATCH_MAX_BYTES)}. ` +
'Remove large files (model weights, datasets, archives, build output) and regenerate the patch; ' +
'use a tiny dummy fixture or a stub instead.'
);
return report;
}
if (report.bytes >= PATCH_WARN_BYTES) {
report.warnings.push(
`environment/workspace.patch is ${fmt(report.bytes)} (limit ${fmt(PATCH_MAX_BYTES)}) — ` +
'check it contains only what the task needs'
);
}
for (const f of largeBinaryFiles(readFileSync(patchPath, 'latin1'))) {
report.warnings.push(
`workspace.patch adds a ${fmt(f.bytes)} binary file: ${f.path} — ` +
'prefer a tiny dummy fixture (e.g. a small random-init checkpoint) or a stub'
);
}
return report;
}

View File

@@ -151,6 +151,25 @@ def _scan_stream_json(stream_path: Path) -> set[str]:
return found
def resolve_capture_dir(reference_run_dir: Path | str) -> Path:
"""The directory whose ``agent/trajectory.json`` is the run's captured transcript.
``reference-runs/<run>/agent/`` is gitignored, so a checkout carries no
trajectory there; the stored atomic regrade at
``rubric-regrades/<run>/agent/trajectory.json`` is the same transcript
(copy-reference-run files the replay trial verbatim). Prefer the run's own
copy, fall back to the stored regrade's, and return the run dir itself when
neither exists so callers report the canonical path.
"""
ref = Path(reference_run_dir)
if (ref / "agent" / "trajectory.json").exists():
return ref
stored = ref.parent.parent / "rubric-regrades" / ref.name
if (stored / "agent" / "trajectory.json").exists():
return stored
return ref
def captured_mutating_tools(reference_run_dir: Path | str) -> set[str]:
"""Return the file-mutating tool names found in a captured run's transcript.

View File

@@ -40,7 +40,12 @@ _scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# .bashrc (the key, the call origin) nor the profile that puts the CLI on PATH.
# Failures stay swallowed — an unreadable .env must not stop the agent starting.
export PATH="$HOME/.local/bin:$PATH"
if [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$_scripts_dir" . "$_scripts_dir/lib/harness-credentials.sh" 2>/dev/null || true
if command -v harness_load_env >/dev/null 2>&1; then
harness_load_env || true
elif [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
# Untrimmed, but a key with a stray \r beats no key at all.
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true

View File

@@ -5,8 +5,8 @@ its captured workspace state — no model calls, no agent work.
Reads a `reference_run_dir` pointing at a `reference-runs/<id>/` directory
captured during a prior real trial. Inside that dir:
agent/trajectory.json — the grader reads this via the symlink
/tmp/outputs/task_transcript.txt → /logs/agent/trajectory.json
agent/trajectory.json — the grader reads this through the transcript
symlink test.sh creates (→ /logs/agent/trajectory.json)
agent-output/ — files the agent created or modified (captured by
tests/test.sh after the agent ran)
agent-output/_HARBOR_DELETIONS.txt
@@ -44,12 +44,13 @@ from harbor.models.trial.paths import EnvironmentPaths
# Sibling module (shipped alongside in the worker toolkit; on PYTHONPATH via
# harbor-regrade). Kept harbor-free so its logic stays unit-testable.
from reference_run_capture import captured_mutating_tools
from reference_run_capture import captured_mutating_tools, resolve_capture_dir
_WORKSPACE = PurePosixPath("/workspace")
_DELETIONS_MARKER = "_HARBOR_DELETIONS.txt"
class ReplayAgent(BaseAgent):
"""Replays a captured reference run so the verifier can be re-graded
without invoking the model again."""
@@ -82,7 +83,15 @@ class ReplayAgent(BaseAgent):
raise FileNotFoundError(f"reference_run_dir does not exist: {ref}")
self._reference_run_dir = ref
self._agent_output_dir = ref / "agent-output"
self._trajectory_path = ref / "agent" / "trajectory.json"
self._capture_dir = resolve_capture_dir(ref)
self._trajectory_path = self._capture_dir / "agent" / "trajectory.json"
if self._capture_dir != ref:
self.logger.info(
"reference_run %s has no agent/trajectory.json; replaying the "
"stored regrade's copy at %s",
ref,
self._trajectory_path,
)
@staticmethod
def name() -> str:
@@ -105,12 +114,12 @@ class ReplayAgent(BaseAgent):
# workspace edits, so a faithful capture of one has an empty (or, in
# older pipelines, absent) agent-output/. That is not a data gap: the
# deliverable is the agent's final message, captured in
# agent/trajectory.json, which the grader reads via
# /tmp/outputs/task_transcript.txt. So overlay captured edits when
# agent/trajectory.json, which the grader reads through test.sh's
# transcript symlink. So overlay captured edits when
# present; otherwise grade the base workspace + transcript, exactly
# what the original advisory grading saw.
if not self._agent_output_dir.is_dir():
mutating = captured_mutating_tools(self._reference_run_dir)
mutating = captured_mutating_tools(self._capture_dir)
if mutating:
# The agent edited files but they weren't captured — grading the
# base workspace would silently score the wrong state. Refuse.
@@ -160,12 +169,12 @@ class ReplayAgent(BaseAgent):
)
# 3. Materialize the captured trajectory at the path the grader's
# test.sh symlinks to /tmp/outputs/task_transcript.txt.
# test.sh symlinks its transcript to.
await self._upload_trajectory(environment)
async def _upload_trajectory(self, environment: BaseEnvironment) -> None:
"""Upload agent/trajectory.json to the path the grader's test.sh
symlinks to /tmp/outputs/task_transcript.txt. Shared by the normal
symlinks its transcript to. Shared by the normal
(overlay) path and the advisory (no agent-output) path."""
if self._trajectory_path.exists():
env_paths = EnvironmentPaths.for_os(environment.os)
@@ -174,7 +183,7 @@ class ReplayAgent(BaseAgent):
target_path=str(env_paths.agent_dir / "trajectory.json"),
)
else:
# The grader's test.sh reads /tmp/outputs/task_transcript.txt,
# The grader's test.sh reads its transcript symlink,
# which symlinks to trajectory.json. Without the file the symlink
# dangles and the grader sees an empty transcript — so the regrade
# will look like the agent did nothing. Yell via harbor's own

View File

@@ -156,7 +156,12 @@ set -euo pipefail
# ~/.local/bin, where the CLI itself lives. Both are set here so a launch works the
# same either way, with the key .env holds right now.
export PATH="\$HOME/.local/bin:\$PATH"
if [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
# Through the lib, not a bare source: the key a custom codex provider authenticates with
# is this env var, and a .env saved on Windows leaves a \\r on it that the proxy 401s.
HARNESS_SCRIPTS_DIR="$_HARNESS_REGISTRY_DIR" . "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh" 2>/dev/null || true
if command -v harness_load_env >/dev/null 2>&1; then
harness_load_env || true
elif [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
. "\${RACCOON_ENV_FILE:-/workspace/.env}"
set +a

View File

@@ -24,7 +24,7 @@ import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
import { holisticRubricScaffoldFor } from './holistic-rubric-scaffold';
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
@@ -234,6 +234,7 @@ mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{ src: 'codex-grader.py', dest: 'tests/codex-grader.py' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
@@ -663,6 +664,7 @@ gpus = 0
allow_internet = true
[verifier.env]
GRADER_HARNESS = "codex"
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
@@ -764,7 +766,7 @@ const holisticRubricMd = `<!--
when you are done.
-->
${HOLISTIC_RUBRIC_SCAFFOLD}`;
${holisticRubricScaffoldFor(slug)}`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');

View File

@@ -24,11 +24,11 @@ from pathlib import Path
import atif_session
import browser_note
try:
from dnsjail import apply_dns_jail
from dnsjail import jailed
except ImportError: # no helper shipped -> no jail, rather than no trials
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
return None
def jailed(run): # type: ignore[misc]
return run
from harbor.agents.installed.base import CliFlag
from harbor.agents.installed.claude_code import ClaudeCode
@@ -208,10 +208,10 @@ class PreinstalledClaudeCode(ClaudeCode):
def name() -> str:
return "claude-code-reduced-toolset"
# Manual tasks inherit harbor's run(), so the jail has to be applied here as well as on
# Manual tasks inherit harbor's run(), so this override needs its own @jailed as well as
# the snapshot path — which overrides run() and never reaches this one.
@jailed
async def run(self, instruction, environment, context) -> None: # type: ignore[override]
await apply_dns_jail(self, environment)
await super().run(instruction, environment, context)
def __init__(self, *args, **kwargs) -> None:
@@ -297,13 +297,36 @@ class PreinstalledClaudeCode(ClaudeCode):
(probe.stdout or "").strip(),
)
return
_log.info(
"image bakes claude %s but %s was requested; using stock installer",
baked,
pinned,
)
_log.info("image bakes claude %s but %s was requested; installing it", baked, pinned)
if await self._install_claude_version(environment, pinned):
return
await ClaudeCode.install(self, environment)
async def _install_claude_version(self, environment, version: str) -> bool:
"""Install an exact claude with install.sh alone, which needs only curl. The stock
installer apt-installs system deps first, and that fails where apt sources have rotted."""
try:
result = await environment.exec(
command=(
f"curl -fsSL https://claude.ai/install.sh | bash -s {shlex.quote(version)} >&2 && "
'export PATH="$HOME/.local/bin:$PATH" && claude --version'
),
timeout_sec=300,
)
except Exception as exc: # noqa: BLE001 — a transport failure falls back like the probe's
_log.warning("install.sh %s raised (%s); falling back to the stock installer", version, exc)
return False
installed = self.parse_version(result.stdout or "") if result.return_code == 0 else None
if installed != version:
_log.warning(
"install.sh %s gave %r (rc=%s); falling back to the stock installer",
version,
installed,
result.return_code,
)
return False
return True
def build_cli_flags(self) -> str:
"""Emit the reduced ``bash + str_replace_editor`` toolset flags: restrict
the built-in toolset to ``--tools Bash`` and append the toolset note.
@@ -477,6 +500,7 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
def _is_bedrock_mode() -> bool:
return False
@jailed
async def run(self, instruction: str, environment, context) -> None:
env = self._build_env()
config_dir = env["CLAUDE_CONFIG_DIR"]
@@ -561,7 +585,6 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
"Seeded session.jsonl is empty; skipping --resume and starting fresh"
)
await apply_dns_jail(self, environment)
await self.exec_as_agent(
environment,
command=(

View File

@@ -18,14 +18,14 @@
* npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>
* npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]
*
* Ships in the worker toolkit (see raccoon-worker-toolkit/package-worker-toolkit.ts),
* Ships in the worker toolkit (see build/package-worker-toolkit.ts),
* so it must only import from its shipped file set — same constraint as
* copy-reference-run.ts.
*/
import './lib/check-devcontainer';
import { existsSync, readFileSync, readdirSync, statSync, writeFileSync } from 'fs';
import { existsSync, readFileSync, readdirSync, realpathSync, statSync, writeFileSync } from 'fs';
import { fileURLToPath } from 'node:url';
import { basename, join, resolve } from 'path';
@@ -126,8 +126,16 @@ export function applyCommand(captureFile: string, jobDirs: string[]): number {
return stamped;
}
// Main. Guarded so the test file can import the commands without running them.
if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1])) {
// Main. Guarded so the test file can import the commands without running them. Real paths
// on both sides: a repo checkout invokes this file through a symlink.
const isMain = (() => {
try {
return !!process.argv[1] && realpathSync(resolve(process.argv[1])) === realpathSync(fileURLToPath(import.meta.url));
} catch {
return false;
}
})();
if (isMain) {
const [command, ...rest] = process.argv.slice(2);
if (command === 'capture') {
const outIdx = rest.indexOf('--out');

View File

@@ -23,6 +23,7 @@ import {
diffTaskInputs,
readTaskInputChecksums,
} from './lib/input-checksums';
import { checkWorkspacePatchSize } from './lib/patch-size';
import {
bannerize,
checkTaskInfraIntegrity,
@@ -94,6 +95,12 @@ export interface TaskStalenessReport {
generatedAt: string;
referenceRuns: ReferenceRunStaleness[];
detectors: DetectorReportStaleness[];
/**
* Detector reports that name reference runs no longer present under
* reference-runs/ — a regrade adopted with --replace, or a re-copied trial,
* re-minted the run folder after the report was written. Absent when none.
*/
detectorRunReferences?: Array<{ report: string; missingRuns: string[] }>;
}
export interface ValidationResult {
@@ -522,13 +529,13 @@ export function validateTask(
// (fetch-submission and the review pipeline both pass the repo root as cwd),
// where .claude/skills additionally holds pipeline-only detectors that
// workers cannot run and must never be warned about.
// raccoon-worker-toolkit/static/.claude/skills is therefore preferred;
// in a packed toolkit only .claude/skills exists, and it IS the shipped set.
// toolkit/static/.claude/skills (the toolkit submodule's shipped set) is therefore
// preferred; in a packed toolkit only .claude/skills exists, and it IS the shipped set.
// Missing reports stay a warning, not an error: the package is reviewable
// without them, but a reviewer has to regenerate the set by hand, which stalls
// the queue. A zero-byte report counts as missing — a stub carries no verdict.
const detectorSkillsDir = [
join(cwd, 'raccoon-worker-toolkit', 'static', '.claude', 'skills'),
join(cwd, 'toolkit', 'static', '.claude', 'skills'),
join(cwd, '.claude', 'skills'),
].find((p) => existsSync(p));
const expectedDetectors = detectorSkillsDir
@@ -795,6 +802,36 @@ export function validateTask(
: [];
const refCount = runDirs.length;
// Detector reports that name runs no longer in the package. The run-reading
// detectors key their findings on reference-run folder names
// (reward-<r>-<id>), and a holistic regrade adopted with --replace re-mints
// that folder under the new reward and trial id — so a report written before
// the regrade describes grades and runs that are gone. Input stamps cannot
// see this (the run set is not a stamped input), so compare the names each
// report cites against the directories that exist.
if (existsSync(detectorsDir) && runDirs.length > 0) {
const runToken = /\breward-\d+(?:\.\d+)?-[A-Za-z0-9][A-Za-z0-9_-]*/g;
const present = new Set(runDirs);
const orphaned: Array<{ report: string; missingRuns: string[] }> = [];
for (const report of readdirSync(detectorsDir).filter((f) => f.endsWith('.md'))) {
const cited = new Set(readFileSync(join(detectorsDir, report), 'utf8').match(runToken) ?? []);
const missing = [...cited].filter((id) => !present.has(id)).sort();
if (missing.length > 0) orphaned.push({ report, missingRuns: missing });
}
if (orphaned.length > 0) {
staleness.detectorRunReferences = orphaned;
const detail = orphaned.map((o) => `${o.report} → ${o.missingRuns.join(', ')}`).join('; ');
logger.warn(
{ orphaned },
`These detector reports name reference runs that are no longer in reference-runs/ — a regrade adopted with --replace, or a re-copied trial, re-minted the run folder after the report was written: ${detail}. Re-run those detectors against the current runs and re-package.`
);
warnings.push(
`${orphaned.length} detector report(s) name reference runs no longer in reference-runs/ — ${detail} — ` +
're-run them against the current runs'
);
}
}
const scores: number[] = [];
const correctnessScores: number[] = [];
const RECOMMENDED_REF_COUNT = 4;
@@ -962,6 +999,12 @@ export function validateTask(
);
}
const patchSize = checkWorkspacePatchSize(join(taskDir, 'environment', 'workspace.patch'));
for (const e of patchSize.errors) logger.error(e);
for (const w of patchSize.warnings) logger.warn(w);
errors.push(...patchSize.errors);
warnings.push(...patchSize.warnings);
// Workspace
const workspaceDir = join(taskDir, 'environment', 'workspace');
let workspaceFiles = 0;