after moving all to cipher

This commit is contained in:
2026-08-19 10:19:57 +00:00
parent 4df62d2609
commit 9abade1a81
100 changed files with 1286 additions and 4335 deletions

View File

@@ -0,0 +1,53 @@
"""Shared browser-capability disclosure for the agent harnesses.
Only images for browser-facing repos ship Playwright, so the note is conditional on probing
the sandbox for the `pw` wrapper rather than on anything about the task. Probing keeps the
claim true by construction: telling an agent it has a browser it does not have sends it after
a missing binary. To check an image yourself: `command -v pw`.
Both harnesses disclose the same text through their own mechanism:
- Claude Code: appended to --append-system-prompt (scripts/snapshot_agent.py)
- codex: -c developer_instructions=... (scripts/codex_agent.py), which prepends a
developer message and LEAVES codex's base instructions intact. Verified with
`codex debug prompt-input`. Do not switch to model_instructions_file — that
REPLACES the base instructions.
This module exists so the probe and the text live in one place; a copy in each adapter would
drift and the drift would be invisible (both would still run, just disclosing differently).
"""
from __future__ import annotations
import logging
from pathlib import Path
_log = logging.getLogger(__name__)
_NOTE_FILE = Path(__file__).resolve().parent / "toolset_note_browser.md"
_PROBE = "command -v pw >/dev/null 2>&1 && echo yes || echo no"
def browser_note() -> str:
"""The disclosure text, or "" if the note file is missing (never fatal)."""
try:
return _NOTE_FILE.read_text(encoding="utf-8").strip()
except OSError:
_log.warning("%s missing; browser note omitted", _NOTE_FILE.name)
return ""
async def probe_browser(environment) -> bool:
"""True when this image ships the `pw` wrapper. Best-effort: a failed probe means no
note, never a failed run."""
try:
result = await environment.exec(command=_PROBE, timeout_sec=30)
except Exception as exc:
_log.warning("browser probe failed (%s); omitting the browser note", exc)
return False
# Exact tail match, not a substring: several harbor environments exec through a LOGIN
# shell, whose profile scripts can print to stdout. A banner containing "yes" would
# otherwise claim a browser that isn't there — the precise failure this module exists
# to prevent.
found = (getattr(result, "stdout", "") or "").strip().endswith("yes")
_log.info("browser probe: pw %s", "present" if found else "absent")
return found

View File

@@ -62,6 +62,33 @@ WORKSPACE="$TASK_DIR/environment/workspace"
echo "Building workspace for $TASK_SLUG"
echo " Commit: $COMMIT"
# `browser = true` in task.toml gives the trial Playwright + Chromium. The build has no way
# to read task.toml — a Dockerfile can only see its build context — so the answer is written
# here as a file the Dockerfile COPYs.
#
# ALWAYS write it, including the "0" case: the COPY is unconditional, and a missing source
# fails the build. Accepts `true` and `"true"`, since the quoted form is a plausible hand-edit
# and rejecting it would silently give a task no browser after its author asked for one.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
mkdir -p "$TASK_DIR/environment"
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
# Not every member's image ships a browser, and the Explore container has one either way — so
# a task can ask for a browser it will not get. Say so here rather than let it pass silently.
if [ "$BROWSER_OPTIN" = "1" ]; then
if grep -q "COPY browser-optin" "$TASK_DIR/environment/Dockerfile" 2>/dev/null; then
echo " Browser: Playwright + Chromium (browser = true)"
else
echo " WARNING: browser = true, but this task's Dockerfile has no browser. The agent" >&2
echo " will get the Read tool and no Chromium. Either drop the flag, or use a" >&2
echo " member whose image ships one:" >&2
echo " grep -l 'COPY browser-optin' task-shared/Dockerfile.*" >&2
fi
fi
# Resolve commit
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
echo " Resolved SHA: $RESOLVED_SHA"

View File

@@ -132,9 +132,13 @@ git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
# `.zeta-siblings/` is staged INTO the workspace by build-workspace.sh on some
# toolkits (bundled sibling deps) — a build artifact, never part of the patch.
# `.raccoon-setup-done` is run-app's per-repo first-use setup marker (polyglot
# toolkits) — authoring-machine state, never task content. run-app git-ignores
# it via the repo's .git/info/exclude, but this script diffs through a
# throwaway --git-dir that never reads that file, so exclude it here too.
# The leading `.` positive pathspec is load-bearing: several git commands
# reject a pathspec made of nothing but exclusions.
EXCLUDES=("." ":(exclude).zeta-siblings")
EXCLUDES=("." ":(exclude).zeta-siblings" ":(exclude).raccoon-setup-done")
cd "$WORKSPACE"
export GIT_WORK_TREE="$WORKSPACE"

View File

@@ -27,6 +27,7 @@ import uuid
from pathlib import Path
import atif_session
import browser_note
from harbor.agents.installed.codex import Codex
from harbor.models.trial.paths import EnvironmentPaths
@@ -77,8 +78,12 @@ _INSTALL_CMD = (
class SystemNodeCodex(Codex):
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
_has_browser = False
async def install(self, environment) -> None: # type: ignore[override]
await self.exec_as_root(environment, command=_INSTALL_CMD)
self._has_browser = await browser_note.probe_browser(environment)
def build_cli_flags(self) -> str: # type: ignore[override]
"""Harbor's flags plus the registry's `agent_config`, so a trial's toolset
@@ -86,7 +91,24 @@ class SystemNodeCodex(Codex):
$RACCOON_AGENT_FLAGS. Both run paths go through here."""
flags = super().build_cli_flags()
reductions = load_harness_registry().require("codex").agent_config_flags()
return f"{flags} {reductions}".strip() if reductions else flags
if reductions:
flags = f"{flags} {reductions}".strip()
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
def _browser_flag(self) -> str:
"""Disclose the browser to codex the way codex takes extra instructions.
`developer_instructions` PREPENDS a developer message and leaves codex's own base
instructions in place — verified with `codex debug prompt-input`. That makes it the
equivalent of claude's --append-system-prompt. `model_instructions_file`, the other
instruction-shaped key, REPLACES the base instructions; do not use it here.
"""
if not self._has_browser:
return ""
note = browser_note.browser_note()
if not note:
return ""
return f"-c developer_instructions={shlex.quote(note)}"
# Where harbor's run-prep stages the prior Claude Code session for snapshot tasks

View File

@@ -64,7 +64,7 @@ import {
INPUT_CHECKSUMS_FILENAME,
readTaskInputChecksums,
} from './lib/input-checksums';
import { didRepair, manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
import { manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
import { readSessionId } from './session-id';
/**
@@ -169,6 +169,27 @@ function copyTrial(trialPath: string, destName?: string) {
process.exit(1);
}
// Repair the SOURCE before reading a byte of it. A trial can leave files
// write-only, which locks out their own owner: everything below — reading
// reward.txt, copying agent-output — fails on them, and any that do get
// through land in the task dir, where harbor hashes every file on every
// later trial and one unreadable path aborts the run.
let sourcePerms = null;
try {
sourcePerms = normalizeTreePermissions(trialPath);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${trialPath}: ${String(err)}`);
console.warn(` If the copy below fails on permissions:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
if (sourcePerms && sourcePerms.failures.length > 0) {
console.warn(
`Warning: could not normalize permissions on ${sourcePerms.failures.length} path(s) under ${trialPath}.`
);
console.warn(` If the copy below fails on permissions, run:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
const rewardPath = join(trialPath, 'verifier', 'reward.txt');
if (!existsSync(rewardPath)) {
console.error(`Error: No reward.txt found in ${trialPath}/verifier/`);
@@ -366,10 +387,10 @@ function copyTrial(trialPath: string, destName?: string) {
console.log(` reward: ${reward}`);
console.log(` task: ${taskDir}`);
console.log(` trial: ${trialId}`);
if (perms && didRepair(perms)) {
console.log(
` perms: normalized ${perms.ownerFixed.length} owner / ${perms.modeFixed.length} mode`
);
const ownerFixed = (sourcePerms?.ownerFixed.length ?? 0) + (perms?.ownerFixed.length ?? 0);
const modeFixed = (sourcePerms?.modeFixed.length ?? 0) + (perms?.modeFixed.length ?? 0);
if (ownerFixed > 0 || modeFixed > 0) {
console.log(` perms: normalized ${ownerFixed} owner / ${modeFixed} mode`);
}
}

View File

@@ -88,6 +88,22 @@ if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ];
exit 1
fi
# Preflight: recompute the browser marker from task.toml.
#
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
# by leaving it alone.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
if [ -d "$TASK_DIR/environment" ]; then
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
fi
# Preflight: report — never block — on edits to toolkit-managed files.
#
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt.md ship from

View File

@@ -4,6 +4,11 @@ version = 1
id = "claude-code"
label = "Claude Code"
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
# given a browser can look at the screenshot it just took. Distinct classes with distinct
# names, because a different toolset is a different agent.
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
import_path_aliases = [
"snapshot_agent:FullToolsetSnapshotClaudeCode",
@@ -29,7 +34,7 @@ install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash &&
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
# unlike model and effort, which are interpolated from this row.
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools Bash --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
[[harness]]
id = "codex"
@@ -84,7 +89,7 @@ explore_config = """
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
"""
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
[[harness]]
id = "gemini-cli"

View File

@@ -36,6 +36,11 @@ class Harness:
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
# has no `Read` equivalent to switch toolsets for.
agent_import_path_browser: str | None = None
agent_import_path_single_turn_browser: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
@@ -67,9 +72,22 @@ class Harness:
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool) -> str:
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there."""
single-turn task given it would try to resume a session that isn't there.
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
built-in — a different toolset is a different agent, so it is a different class with
its own name rather than a flag on the canonical one. Harnesses without a variant fall
through to their normal class."""
if browser:
variant = (
self.agent_import_path_browser
if multi_turn
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
)
if variant:
return variant
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
@@ -180,6 +198,8 @@ _KNOWN_FIELDS = frozenset(
"label",
"agent_import_path",
"agent_import_path_single_turn",
"agent_import_path_browser",
"agent_import_path_single_turn_browser",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
@@ -315,6 +335,8 @@ def _build(entry: dict, index: int) -> Harness:
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
agent_import_path_browser=entry.get("agent_import_path_browser"),
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),

View File

@@ -6,10 +6,21 @@
* *inside* it, so the repair has to fix directory modes, not just ownership.
* These tests run unprivileged, so they exercise the mode axis for real and the
* ownership axis only as far as an unprivileged process can (target resolution +
* graceful EPERM), which is the same shape CI runs in.
* graceful EPERM), which is the same shape CI runs in. One case needs root and
* skips otherwise; the rest hold under either uid, which is why the fixtures that
* must look human-owned say so with `ownedByHuman` instead of relying on the
* caller's uid.
*/
import assert from 'node:assert/strict';
import { chmodSync, mkdirSync, rmSync, statSync, symlinkSync, writeFileSync } from 'node:fs';
import {
chmodSync,
chownSync,
mkdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { test } from 'node:test';
@@ -28,6 +39,16 @@ function scratch(name: string): string {
return dir;
}
const RUNNING_AS_ROOT = process.getuid?.() === 0;
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
function ownedByHuman(path: string): string {
chownSync(path, HUMAN_UID, HUMAN_GID);
return path;
}
test('restores the search bit on a directory that lost it', () => {
const root = scratch('searchbit');
const models = join(root, 'agent-output', 'app', 'models');
@@ -77,12 +98,13 @@ test('leaves already-correct trees untouched', () => {
});
test('does not widen group/other beyond what was already there', () => {
const root = scratch('narrow');
const root = ownedByHuman(scratch('narrow'));
const f = join(root, 'secret.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o000);
normalizeTreePermissions(root);
normalizeTreePermissions(root, { ownerRef: root });
const mode = statSync(f).mode & 0o777;
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
@@ -155,6 +177,47 @@ test('still normalizes modes when the chown target is root', () => {
rmSync(root, { recursive: true, force: true });
});
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
// The complement of the case below: we declined to chown, but these entries are
// already the human's, so owner bits reach them and nothing should be widened.
const root = ownedByHuman(scratch('root-ref-narrow'));
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o600);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
rmSync(root, { recursive: true, force: true });
});
test(
'grants read+search to group and other on files stranded root-owned',
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
() => {
// The worker authoring container: root process, root-owned workspace. The chown
// is declined, so owner bits land on root and the human — a different uid in
// Explore and on a WSL host — is still locked out of a --w------- capture.
const root = scratch('stranded');
const sub = join(root, 'agent-output');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'answer.md');
writeFileSync(f, 'x\n');
chmodSync(f, 0o200);
chmodSync(sub, 0o300);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(
statSync(f).mode & 0o777,
0o644,
'file readable by everyone, writable by none but root'
);
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
}
);
test('walks a tree as deep as the filesystem allows', () => {
const root = scratch('deep');
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name

View File

@@ -34,9 +34,16 @@ export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: num
}
}
/** Owner-rwX mode, preserving every other bit. Dirs also need the search bit. */
function withOwnerAccess(mode: number, isDir: boolean): number {
return mode | (isDir ? 0o700 : 0o600);
/**
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
*
* `stranded` means the file stays root-owned because we have no non-root owner to
* give it to. Owner bits then help nobody — whoever has to read it is a different
* user — so read and search are granted more widely. Never write, never +x on files.
*/
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
const owner = isDir ? 0o700 : 0o600;
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
}
/**
@@ -75,7 +82,7 @@ export function normalizeTreePermissions(
const isDir = st.isDirectory();
// Mode first: a directory we can't search is one we can't descend into.
const wanted = withOwnerAccess(st.mode, isDir);
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
if (wanted !== st.mode) {
try {
chmodSync(path, wanted);

View File

@@ -36,6 +36,7 @@ from __future__ import annotations
import argparse
import json
import os
import re
import shlex
import sys
import tomllib
@@ -77,6 +78,26 @@ def is_multi_turn(task_dir: str | None) -> bool:
return session.is_file() and session.stat().st_size > 0
def wants_browser(task_dir: str | None) -> bool:
"""True when task.toml opts into a browser (`[metadata] browser = true`).
Read straight from the file rather than via tomllib: this must agree with
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
text match. If the two ever disagree the agent is told about a browser the image
lacks, which is the one failure the disclosure is designed to make impossible.
Accepts the quoted form for the same reason build-workspace.sh does."""
if not task_dir:
return False
toml_path = Path(task_dir) / "task.toml"
if not toml_path.is_file():
return False
try:
text = toml_path.read_text(encoding="utf-8")
except OSError:
return False
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
def harness_from_task_toml(task_dir: str | None) -> str | None:
"""The task's own `[agent] harness` — the authoritative record of which harness
this task was authored against.
@@ -388,8 +409,9 @@ def main(argv: list[str] | None = None) -> int:
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
# anything it would only echo back at the worker is said below instead.
browser = wants_browser(args.task_dir)
assignments = {
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn),
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
"MODEL": normalize_model(harness, model),
"EFFORT_KWARG": harness.effort_kwarg,
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
@@ -398,8 +420,17 @@ def main(argv: list[str] | None = None) -> int:
warn(
f"{harness.label} · model={assignments['MODEL']} · "
f"{'multi-turn' if multi_turn else 'single-turn'} · "
f"{'browser · ' if browser else ''}"
f"agent={assignments['AGENT_IMPORT_PATH']}"
)
if browser and not harness.agent_import_path_browser:
# Not a failure: the image still gets Playwright and the agent is still told about
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
# letting someone infer from a log line that the opt-in was ignored entirely.
warn(
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
f"The browser and its disclosure are unaffected."
)
if harness.flaky_hangs:
warn(
f"{harness.label} is known to hang with no client-side timeout on a small "

View File

@@ -113,11 +113,24 @@ harness_install_launchers() {
mkdir -p "$bin"
# Read at launcher run time so the note stays a file, not a baked-in copy.
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
local browser_note_src="${note_src%.md}_browser.md"
local read_note_src="${note_src%.md}_read.md"
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
local id cli launch
local id cli launch switchable
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
# so a browser task needs nothing added and the flag has nothing to switch.
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
# which every launch line references, and every harness would look switchable.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
switchable=1
else
switchable=0
fi
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
@@ -127,7 +140,38 @@ if [ -f "$note_src" ]; then
else
RACCOON_TOOLSET_NOTE=""
fi
export RACCOON_TOOLSET_NOTE
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
# here. Per invocation, not per container — authoring a browser task shouldn't need a
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
# still mirrors an ordinary trial.
#
# The correction must be appended AFTER the base note, which says there is no Read tool.
RACCOON_TOOLS="Bash"
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
RACCOON_TOOLS="Bash,Read"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\$(cat "$read_note_src")"
fi
export RACCOON_TOOLS
# Only mention the browser on an image that actually has one — most don't. Probed at
# launch, not baked in, so the same launcher is correct in whichever container it runs.
#
# Exported two ways because the harnesses take extra instructions differently: claude
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
# (it has no str_replace_editor), so the browser part is exported on its own too.
RACCOON_BROWSER_NOTE=""
RACCOON_BROWSER_FLAGS=()
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\${RACCOON_BROWSER_NOTE}"
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
fi
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
export RACCOON_HARNESS="$id"
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
@@ -143,11 +187,33 @@ LAUNCHER
# Alias lines for ~/.bashrc.
harness_alias_lines() {
local id cli launch
local id cli launch switchable
local browser_clis=""
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
echo "alias $cli=\"raccoon-explore-$cli\""
# Same derivation as the launcher: only a harness whose launch line takes
# $RACCOON_TOOLS has a toolset the flag can change.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
browser_clis="${browser_clis:+$browser_clis }$cli"
fi
done < <(_harness_query --explore-launchers 2>/dev/null || true)
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
# alternate screen buffer, so anything printed just before exec is hidden for the whole
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
#
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
# and in one without.
[ -n "$browser_clis" ] || return 0
local first="${browser_clis%% *}"
cat <<HINT
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
fi
HINT
}
# Write each harness's config file from the registry, replacing whatever was there.

View File

@@ -557,6 +557,9 @@ repo = "${repoName}"
commit = "${commitShort}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]

View File

@@ -22,6 +22,7 @@ import tempfile
from pathlib import Path
import atif_session
import browser_note
from harbor.agents.installed.claude_code import ClaudeCode
from harbor.models.trial.paths import EnvironmentPaths
@@ -101,7 +102,7 @@ _AGENT_CLI_NOTE_FALLBACK = (
)
def _toolset_note() -> str:
def _toolset_note(with_browser: bool = False, with_read: bool = False) -> str:
"""The toolset note appended to Claude Code's stock ``--print`` system prompt
(via ``--append-system-prompt``) for the canonical reduced toolset.
@@ -118,9 +119,23 @@ def _toolset_note() -> str:
file is missing."""
path = Path(__file__).resolve().parent / "toolset_note.md"
try:
return path.read_text(encoding="utf-8").strip()
note = path.read_text(encoding="utf-8").strip()
except OSError:
return _AGENT_CLI_NOTE_FALLBACK
note = _AGENT_CLI_NOTE_FALLBACK
# Order matters: the Read correction must come AFTER the base note, because it supersedes
# that note's "there are no Read/Grep/Glob tools" line. Shipping the base note alone to a
# Read-enabled agent would be a false statement about its own toolset.
if with_read:
read_note = Path(__file__).resolve().parent / "toolset_note_read.md"
try:
note = f"{note}\n\n{read_note.read_text(encoding='utf-8').strip()}"
except OSError:
_log.warning("toolset_note_read.md missing; Read correction omitted")
if with_browser:
extra = browser_note.browser_note()
if extra:
note = f"{note}\n\n{extra}"
return note
class PreinstalledClaudeCode(ClaudeCode):
@@ -138,6 +153,10 @@ class PreinstalledClaudeCode(ClaudeCode):
in the trial config's ``agent.import_path``.)
"""
# Set by _probe_browser() during install(); read by build_cli_flags(). Declared
# here so the full-toolset subclass (which skips the probe) still has a value.
_has_browser = False
@staticmethod
def name() -> str:
return "claude-code-reduced-toolset"
@@ -176,6 +195,14 @@ class PreinstalledClaudeCode(ClaudeCode):
editor CLI) and reuses ``_ensure_claude_binary`` directly."""
await self._stage_agent_cli(environment)
await self._ensure_claude_binary(environment)
await self._probe_browser(environment)
async def _probe_browser(self, environment) -> None:
"""Record whether this image ships the Playwright `pw` wrapper, so the toolset
note mentions the browser only on images that have one. Runs during install(),
which harbor calls before build_cli_flags() reads the result. The probe and the
note text are shared with the codex adapter via browser_note.py."""
self._has_browser = await browser_note.probe_browser(environment)
async def _ensure_claude_binary(self, environment) -> None:
"""Reuse the claude binary already baked into the task image instead
@@ -243,7 +270,8 @@ class PreinstalledClaudeCode(ClaudeCode):
subclass overrides this back to stock ``ClaudeCode.build_cli_flags``.
"""
flags = super().build_cli_flags()
extra = f"--tools Bash --append-system-prompt {shlex.quote(_toolset_note())}"
note = _toolset_note(with_browser=self._has_browser)
extra = f"--tools Bash --append-system-prompt {shlex.quote(note)}"
return f"{flags} {extra}" if flags else extra
async def _claude_format_session_path(self, environment, env, session_uuid: str) -> str:
@@ -845,6 +873,47 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
# once every snapshot session.jsonl is re-recorded in the reduced format.
class _BrowserToolsetMixin:
"""The canonical reduced toolset PLUS the ``Read`` built-in, for tasks that opt into a
browser: a screenshot is only useful to an agent that can look at it, and ``Read`` is what
turns a PNG on disk into an image the model actually sees.
This is a SEPARATE AGENT, not a flag, per the rule that the toolset is chosen by which class
harbor runs and the class name records it — so a benchmark row can never silently compare an
agent that could see against one that couldn't.
Two things to be clear-eyed about:
* ``Read`` is not image-only. It also reads text files, PDFs and notebooks, so these tasks
get back a file-reading built-in the reduced toolset deliberately removes. There is no
narrower built-in; an image-only MCP tool was rejected because the canonical agent avoids
MCP (see the async-MCP startup race note at the top of this file).
* Tasks on this agent are not comparable with tasks on the canonical one. That is the point
of the distinct name.
"""
def build_cli_flags(self) -> str:
flags = ClaudeCode.build_cli_flags(self)
note = _toolset_note(with_browser=self._has_browser, with_read=True)
extra = f"--tools Bash,Read --append-system-prompt {shlex.quote(note)}"
return f"{flags} {extra}" if flags else extra
class BrowserPreinstalledClaudeCode(_BrowserToolsetMixin, PreinstalledClaudeCode):
"""Reduced toolset + Read, manual (non-snapshot) tasks."""
@staticmethod
def name() -> str:
return "claude-code-reduced-toolset-browser"
class BrowserSnapshotClaudeCode(_BrowserToolsetMixin, SnapshotClaudeCode):
"""Reduced toolset + Read, snapshot tasks."""
@staticmethod
def name() -> str:
return "snapshot-claude-code-reduced-toolset-browser"
class _FullToolsetMixin:
"""Override the canonical reduced toolset back to Claude Code's stock full
built-in toolset: no str_replace_editor CLI to stage, and no --tools / note.

View File

@@ -926,6 +926,33 @@ if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1
execSync(tarCmd, { stdio: 'pipe' });
};
const taskDir = join('harbor-tasks', slug);
// tar records each file's mode as-is, and this container runs as root — so a
// write-only file (Claude Code writes subagent session records --w-------) is
// archived, not refused, and every later extraction of it is unreadable.
// Normalize before packing; the catch below still repairs what only tar can see.
try {
const pre = normalizeTreePermissions(taskDir);
if (didRepair(pre)) {
log.info(
{ ownerFixed: pre.ownerFixed.length, modeFixed: pre.modeFixed.length },
'Normalized workspace permissions before packaging'
);
}
if (pre.failures.length > 0) {
log.warn(
{ count: pre.failures.length, paths: pre.failures.slice(0, 5).map((f) => f.path) },
`Could not normalize some paths. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}`
);
}
} catch (repairErr) {
log.warn(
{ err: repairErr },
`Permission normalization failed — packaging anyway. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}`
);
}
try {
runTar();
} catch (err) {
@@ -952,7 +979,6 @@ if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1
}
}
const taskDir = join('harbor-tasks', slug);
// Guarded so a failed repair can't mask the real packaging error.
let perms;
try {

View File

@@ -0,0 +1,4 @@
## Browser
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
`require("playwright")` resolvable (CommonJS — `import` will not find it).

View File

@@ -0,0 +1,7 @@
## Correction to the toolset above: you also have `Read`
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
edit files with `str_replace_editor`.