after moving all to cipher
This commit is contained in:
53
worker-toolkit-stocks-in-the-future/scripts/browser_note.py
Normal file
53
worker-toolkit-stocks-in-the-future/scripts/browser_note.py
Normal file
@@ -0,0 +1,53 @@
|
||||
"""Shared browser-capability disclosure for the agent harnesses.
|
||||
|
||||
Only images for browser-facing repos ship Playwright, so the note is conditional on probing
|
||||
the sandbox for the `pw` wrapper rather than on anything about the task. Probing keeps the
|
||||
claim true by construction: telling an agent it has a browser it does not have sends it after
|
||||
a missing binary. To check an image yourself: `command -v pw`.
|
||||
|
||||
Both harnesses disclose the same text through their own mechanism:
|
||||
- Claude Code: appended to --append-system-prompt (scripts/snapshot_agent.py)
|
||||
- codex: -c developer_instructions=... (scripts/codex_agent.py), which prepends a
|
||||
developer message and LEAVES codex's base instructions intact. Verified with
|
||||
`codex debug prompt-input`. Do not switch to model_instructions_file — that
|
||||
REPLACES the base instructions.
|
||||
|
||||
This module exists so the probe and the text live in one place; a copy in each adapter would
|
||||
drift and the drift would be invisible (both would still run, just disclosing differently).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
_log = logging.getLogger(__name__)
|
||||
|
||||
_NOTE_FILE = Path(__file__).resolve().parent / "toolset_note_browser.md"
|
||||
_PROBE = "command -v pw >/dev/null 2>&1 && echo yes || echo no"
|
||||
|
||||
|
||||
def browser_note() -> str:
|
||||
"""The disclosure text, or "" if the note file is missing (never fatal)."""
|
||||
try:
|
||||
return _NOTE_FILE.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_log.warning("%s missing; browser note omitted", _NOTE_FILE.name)
|
||||
return ""
|
||||
|
||||
|
||||
async def probe_browser(environment) -> bool:
|
||||
"""True when this image ships the `pw` wrapper. Best-effort: a failed probe means no
|
||||
note, never a failed run."""
|
||||
try:
|
||||
result = await environment.exec(command=_PROBE, timeout_sec=30)
|
||||
except Exception as exc:
|
||||
_log.warning("browser probe failed (%s); omitting the browser note", exc)
|
||||
return False
|
||||
# Exact tail match, not a substring: several harbor environments exec through a LOGIN
|
||||
# shell, whose profile scripts can print to stdout. A banner containing "yes" would
|
||||
# otherwise claim a browser that isn't there — the precise failure this module exists
|
||||
# to prevent.
|
||||
found = (getattr(result, "stdout", "") or "").strip().endswith("yes")
|
||||
_log.info("browser probe: pw %s", "present" if found else "absent")
|
||||
return found
|
||||
@@ -62,6 +62,33 @@ WORKSPACE="$TASK_DIR/environment/workspace"
|
||||
echo "Building workspace for $TASK_SLUG"
|
||||
echo " Commit: $COMMIT"
|
||||
|
||||
# `browser = true` in task.toml gives the trial Playwright + Chromium. The build has no way
|
||||
# to read task.toml — a Dockerfile can only see its build context — so the answer is written
|
||||
# here as a file the Dockerfile COPYs.
|
||||
#
|
||||
# ALWAYS write it, including the "0" case: the COPY is unconditional, and a missing source
|
||||
# fails the build. Accepts `true` and `"true"`, since the quoted form is a plausible hand-edit
|
||||
# and rejecting it would silently give a task no browser after its author asked for one.
|
||||
BROWSER_OPTIN=0
|
||||
if [ -f "$TASK_DIR/task.toml" ] &&
|
||||
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
|
||||
BROWSER_OPTIN=1
|
||||
fi
|
||||
mkdir -p "$TASK_DIR/environment"
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
|
||||
# Not every member's image ships a browser, and the Explore container has one either way — so
|
||||
# a task can ask for a browser it will not get. Say so here rather than let it pass silently.
|
||||
if [ "$BROWSER_OPTIN" = "1" ]; then
|
||||
if grep -q "COPY browser-optin" "$TASK_DIR/environment/Dockerfile" 2>/dev/null; then
|
||||
echo " Browser: Playwright + Chromium (browser = true)"
|
||||
else
|
||||
echo " WARNING: browser = true, but this task's Dockerfile has no browser. The agent" >&2
|
||||
echo " will get the Read tool and no Chromium. Either drop the flag, or use a" >&2
|
||||
echo " member whose image ships one:" >&2
|
||||
echo " grep -l 'COPY browser-optin' task-shared/Dockerfile.*" >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# Resolve commit
|
||||
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
|
||||
echo " Resolved SHA: $RESOLVED_SHA"
|
||||
|
||||
@@ -132,9 +132,13 @@ git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
|
||||
|
||||
# `.zeta-siblings/` is staged INTO the workspace by build-workspace.sh on some
|
||||
# toolkits (bundled sibling deps) — a build artifact, never part of the patch.
|
||||
# `.raccoon-setup-done` is run-app's per-repo first-use setup marker (polyglot
|
||||
# toolkits) — authoring-machine state, never task content. run-app git-ignores
|
||||
# it via the repo's .git/info/exclude, but this script diffs through a
|
||||
# throwaway --git-dir that never reads that file, so exclude it here too.
|
||||
# The leading `.` positive pathspec is load-bearing: several git commands
|
||||
# reject a pathspec made of nothing but exclusions.
|
||||
EXCLUDES=("." ":(exclude).zeta-siblings")
|
||||
EXCLUDES=("." ":(exclude).zeta-siblings" ":(exclude).raccoon-setup-done")
|
||||
|
||||
cd "$WORKSPACE"
|
||||
export GIT_WORK_TREE="$WORKSPACE"
|
||||
|
||||
@@ -27,6 +27,7 @@ import uuid
|
||||
from pathlib import Path
|
||||
|
||||
import atif_session
|
||||
import browser_note
|
||||
|
||||
from harbor.agents.installed.codex import Codex
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
@@ -77,8 +78,12 @@ _INSTALL_CMD = (
|
||||
|
||||
|
||||
class SystemNodeCodex(Codex):
|
||||
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
|
||||
_has_browser = False
|
||||
|
||||
async def install(self, environment) -> None: # type: ignore[override]
|
||||
await self.exec_as_root(environment, command=_INSTALL_CMD)
|
||||
self._has_browser = await browser_note.probe_browser(environment)
|
||||
|
||||
def build_cli_flags(self) -> str: # type: ignore[override]
|
||||
"""Harbor's flags plus the registry's `agent_config`, so a trial's toolset
|
||||
@@ -86,7 +91,24 @@ class SystemNodeCodex(Codex):
|
||||
$RACCOON_AGENT_FLAGS. Both run paths go through here."""
|
||||
flags = super().build_cli_flags()
|
||||
reductions = load_harness_registry().require("codex").agent_config_flags()
|
||||
return f"{flags} {reductions}".strip() if reductions else flags
|
||||
if reductions:
|
||||
flags = f"{flags} {reductions}".strip()
|
||||
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
|
||||
|
||||
def _browser_flag(self) -> str:
|
||||
"""Disclose the browser to codex the way codex takes extra instructions.
|
||||
|
||||
`developer_instructions` PREPENDS a developer message and leaves codex's own base
|
||||
instructions in place — verified with `codex debug prompt-input`. That makes it the
|
||||
equivalent of claude's --append-system-prompt. `model_instructions_file`, the other
|
||||
instruction-shaped key, REPLACES the base instructions; do not use it here.
|
||||
"""
|
||||
if not self._has_browser:
|
||||
return ""
|
||||
note = browser_note.browser_note()
|
||||
if not note:
|
||||
return ""
|
||||
return f"-c developer_instructions={shlex.quote(note)}"
|
||||
|
||||
|
||||
# Where harbor's run-prep stages the prior Claude Code session for snapshot tasks
|
||||
|
||||
@@ -64,7 +64,7 @@ import {
|
||||
INPUT_CHECKSUMS_FILENAME,
|
||||
readTaskInputChecksums,
|
||||
} from './lib/input-checksums';
|
||||
import { didRepair, manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
|
||||
import { manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
|
||||
import { readSessionId } from './session-id';
|
||||
|
||||
/**
|
||||
@@ -169,6 +169,27 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Repair the SOURCE before reading a byte of it. A trial can leave files
|
||||
// write-only, which locks out their own owner: everything below — reading
|
||||
// reward.txt, copying agent-output — fails on them, and any that do get
|
||||
// through land in the task dir, where harbor hashes every file on every
|
||||
// later trial and one unreadable path aborts the run.
|
||||
let sourcePerms = null;
|
||||
try {
|
||||
sourcePerms = normalizeTreePermissions(trialPath);
|
||||
} catch (err) {
|
||||
console.warn(`Warning: could not normalize permissions on ${trialPath}: ${String(err)}`);
|
||||
console.warn(` If the copy below fails on permissions:`);
|
||||
console.warn(` ${manualRepairHint(trialPath)}`);
|
||||
}
|
||||
if (sourcePerms && sourcePerms.failures.length > 0) {
|
||||
console.warn(
|
||||
`Warning: could not normalize permissions on ${sourcePerms.failures.length} path(s) under ${trialPath}.`
|
||||
);
|
||||
console.warn(` If the copy below fails on permissions, run:`);
|
||||
console.warn(` ${manualRepairHint(trialPath)}`);
|
||||
}
|
||||
|
||||
const rewardPath = join(trialPath, 'verifier', 'reward.txt');
|
||||
if (!existsSync(rewardPath)) {
|
||||
console.error(`Error: No reward.txt found in ${trialPath}/verifier/`);
|
||||
@@ -366,10 +387,10 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
console.log(` reward: ${reward}`);
|
||||
console.log(` task: ${taskDir}`);
|
||||
console.log(` trial: ${trialId}`);
|
||||
if (perms && didRepair(perms)) {
|
||||
console.log(
|
||||
` perms: normalized ${perms.ownerFixed.length} owner / ${perms.modeFixed.length} mode`
|
||||
);
|
||||
const ownerFixed = (sourcePerms?.ownerFixed.length ?? 0) + (perms?.ownerFixed.length ?? 0);
|
||||
const modeFixed = (sourcePerms?.modeFixed.length ?? 0) + (perms?.modeFixed.length ?? 0);
|
||||
if (ownerFixed > 0 || modeFixed > 0) {
|
||||
console.log(` perms: normalized ${ownerFixed} owner / ${modeFixed} mode`);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -88,6 +88,22 @@ if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ];
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Preflight: recompute the browser marker from task.toml.
|
||||
#
|
||||
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
|
||||
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
|
||||
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
|
||||
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
|
||||
# by leaving it alone.
|
||||
BROWSER_OPTIN=0
|
||||
if [ -f "$TASK_DIR/task.toml" ] &&
|
||||
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
|
||||
BROWSER_OPTIN=1
|
||||
fi
|
||||
if [ -d "$TASK_DIR/environment" ]; then
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
|
||||
fi
|
||||
|
||||
# Preflight: report — never block — on edits to toolkit-managed files.
|
||||
#
|
||||
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt.md ship from
|
||||
|
||||
@@ -4,6 +4,11 @@ version = 1
|
||||
id = "claude-code"
|
||||
label = "Claude Code"
|
||||
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
|
||||
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
|
||||
# given a browser can look at the screenshot it just took. Distinct classes with distinct
|
||||
# names, because a different toolset is a different agent.
|
||||
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
|
||||
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
|
||||
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
|
||||
import_path_aliases = [
|
||||
"snapshot_agent:FullToolsetSnapshotClaudeCode",
|
||||
@@ -29,7 +34,7 @@ install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash &&
|
||||
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
|
||||
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
|
||||
# unlike model and effort, which are interpolated from this row.
|
||||
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools Bash --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
|
||||
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
|
||||
|
||||
[[harness]]
|
||||
id = "codex"
|
||||
@@ -84,7 +89,7 @@ explore_config = """
|
||||
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
|
||||
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
|
||||
"""
|
||||
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
|
||||
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
|
||||
|
||||
[[harness]]
|
||||
id = "gemini-cli"
|
||||
|
||||
@@ -36,6 +36,11 @@ class Harness:
|
||||
seed_native: bool
|
||||
seed_atif: bool
|
||||
agent_import_path_single_turn: str | None = None
|
||||
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
|
||||
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
|
||||
# has no `Read` equivalent to switch toolsets for.
|
||||
agent_import_path_browser: str | None = None
|
||||
agent_import_path_single_turn_browser: str | None = None
|
||||
import_path_aliases: tuple[str, ...] = ()
|
||||
legacy_bare_model_rows: bool = False
|
||||
default_model: str | None = None
|
||||
@@ -67,9 +72,22 @@ class Harness:
|
||||
# be edited in lockstep with the schema.
|
||||
extra: dict = field(default_factory=dict, compare=False)
|
||||
|
||||
def agent_import_path_for(self, *, multi_turn: bool) -> str:
|
||||
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
|
||||
"""Agent class to launch. Multi-turn tasks need the resuming class; a
|
||||
single-turn task given it would try to resume a session that isn't there."""
|
||||
single-turn task given it would try to resume a session that isn't there.
|
||||
|
||||
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
|
||||
built-in — a different toolset is a different agent, so it is a different class with
|
||||
its own name rather than a flag on the canonical one. Harnesses without a variant fall
|
||||
through to their normal class."""
|
||||
if browser:
|
||||
variant = (
|
||||
self.agent_import_path_browser
|
||||
if multi_turn
|
||||
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
|
||||
)
|
||||
if variant:
|
||||
return variant
|
||||
if multi_turn:
|
||||
return self.agent_import_path
|
||||
return self.agent_import_path_single_turn or self.agent_import_path
|
||||
@@ -180,6 +198,8 @@ _KNOWN_FIELDS = frozenset(
|
||||
"label",
|
||||
"agent_import_path",
|
||||
"agent_import_path_single_turn",
|
||||
"agent_import_path_browser",
|
||||
"agent_import_path_single_turn_browser",
|
||||
"import_path_aliases",
|
||||
"legacy_bare_model_rows",
|
||||
"default_model",
|
||||
@@ -315,6 +335,8 @@ def _build(entry: dict, index: int) -> Harness:
|
||||
label=entry["label"],
|
||||
agent_import_path=entry["agent_import_path"],
|
||||
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
|
||||
agent_import_path_browser=entry.get("agent_import_path_browser"),
|
||||
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
|
||||
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
|
||||
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
|
||||
default_model=entry.get("default_model"),
|
||||
|
||||
@@ -6,10 +6,21 @@
|
||||
* *inside* it, so the repair has to fix directory modes, not just ownership.
|
||||
* These tests run unprivileged, so they exercise the mode axis for real and the
|
||||
* ownership axis only as far as an unprivileged process can (target resolution +
|
||||
* graceful EPERM), which is the same shape CI runs in.
|
||||
* graceful EPERM), which is the same shape CI runs in. One case needs root and
|
||||
* skips otherwise; the rest hold under either uid, which is why the fixtures that
|
||||
* must look human-owned say so with `ownedByHuman` instead of relying on the
|
||||
* caller's uid.
|
||||
*/
|
||||
import assert from 'node:assert/strict';
|
||||
import { chmodSync, mkdirSync, rmSync, statSync, symlinkSync, writeFileSync } from 'node:fs';
|
||||
import {
|
||||
chmodSync,
|
||||
chownSync,
|
||||
mkdirSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
symlinkSync,
|
||||
writeFileSync,
|
||||
} from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { test } from 'node:test';
|
||||
@@ -28,6 +39,16 @@ function scratch(name: string): string {
|
||||
return dir;
|
||||
}
|
||||
|
||||
const RUNNING_AS_ROOT = process.getuid?.() === 0;
|
||||
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
|
||||
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
|
||||
|
||||
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
|
||||
function ownedByHuman(path: string): string {
|
||||
chownSync(path, HUMAN_UID, HUMAN_GID);
|
||||
return path;
|
||||
}
|
||||
|
||||
test('restores the search bit on a directory that lost it', () => {
|
||||
const root = scratch('searchbit');
|
||||
const models = join(root, 'agent-output', 'app', 'models');
|
||||
@@ -77,12 +98,13 @@ test('leaves already-correct trees untouched', () => {
|
||||
});
|
||||
|
||||
test('does not widen group/other beyond what was already there', () => {
|
||||
const root = scratch('narrow');
|
||||
const root = ownedByHuman(scratch('narrow'));
|
||||
const f = join(root, 'secret.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o000);
|
||||
|
||||
normalizeTreePermissions(root);
|
||||
normalizeTreePermissions(root, { ownerRef: root });
|
||||
|
||||
const mode = statSync(f).mode & 0o777;
|
||||
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
|
||||
@@ -155,6 +177,47 @@ test('still normalizes modes when the chown target is root', () => {
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
|
||||
// The complement of the case below: we declined to chown, but these entries are
|
||||
// already the human's, so owner bits reach them and nothing should be widened.
|
||||
const root = ownedByHuman(scratch('root-ref-narrow'));
|
||||
const f = join(root, 'mine.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o600);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test(
|
||||
'grants read+search to group and other on files stranded root-owned',
|
||||
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
|
||||
() => {
|
||||
// The worker authoring container: root process, root-owned workspace. The chown
|
||||
// is declined, so owner bits land on root and the human — a different uid in
|
||||
// Explore and on a WSL host — is still locked out of a --w------- capture.
|
||||
const root = scratch('stranded');
|
||||
const sub = join(root, 'agent-output');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
const f = join(sub, 'answer.md');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o200);
|
||||
chmodSync(sub, 0o300);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(
|
||||
statSync(f).mode & 0o777,
|
||||
0o644,
|
||||
'file readable by everyone, writable by none but root'
|
||||
);
|
||||
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
|
||||
}
|
||||
);
|
||||
|
||||
test('walks a tree as deep as the filesystem allows', () => {
|
||||
const root = scratch('deep');
|
||||
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
|
||||
|
||||
@@ -34,9 +34,16 @@ export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: num
|
||||
}
|
||||
}
|
||||
|
||||
/** Owner-rwX mode, preserving every other bit. Dirs also need the search bit. */
|
||||
function withOwnerAccess(mode: number, isDir: boolean): number {
|
||||
return mode | (isDir ? 0o700 : 0o600);
|
||||
/**
|
||||
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
|
||||
*
|
||||
* `stranded` means the file stays root-owned because we have no non-root owner to
|
||||
* give it to. Owner bits then help nobody — whoever has to read it is a different
|
||||
* user — so read and search are granted more widely. Never write, never +x on files.
|
||||
*/
|
||||
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
|
||||
const owner = isDir ? 0o700 : 0o600;
|
||||
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -75,7 +82,7 @@ export function normalizeTreePermissions(
|
||||
const isDir = st.isDirectory();
|
||||
|
||||
// Mode first: a directory we can't search is one we can't descend into.
|
||||
const wanted = withOwnerAccess(st.mode, isDir);
|
||||
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
|
||||
if (wanted !== st.mode) {
|
||||
try {
|
||||
chmodSync(path, wanted);
|
||||
|
||||
@@ -36,6 +36,7 @@ from __future__ import annotations
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shlex
|
||||
import sys
|
||||
import tomllib
|
||||
@@ -77,6 +78,26 @@ def is_multi_turn(task_dir: str | None) -> bool:
|
||||
return session.is_file() and session.stat().st_size > 0
|
||||
|
||||
|
||||
def wants_browser(task_dir: str | None) -> bool:
|
||||
"""True when task.toml opts into a browser (`[metadata] browser = true`).
|
||||
|
||||
Read straight from the file rather than via tomllib: this must agree with
|
||||
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
|
||||
text match. If the two ever disagree the agent is told about a browser the image
|
||||
lacks, which is the one failure the disclosure is designed to make impossible.
|
||||
Accepts the quoted form for the same reason build-workspace.sh does."""
|
||||
if not task_dir:
|
||||
return False
|
||||
toml_path = Path(task_dir) / "task.toml"
|
||||
if not toml_path.is_file():
|
||||
return False
|
||||
try:
|
||||
text = toml_path.read_text(encoding="utf-8")
|
||||
except OSError:
|
||||
return False
|
||||
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
|
||||
|
||||
|
||||
def harness_from_task_toml(task_dir: str | None) -> str | None:
|
||||
"""The task's own `[agent] harness` — the authoritative record of which harness
|
||||
this task was authored against.
|
||||
@@ -388,8 +409,9 @@ def main(argv: list[str] | None = None) -> int:
|
||||
|
||||
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
|
||||
# anything it would only echo back at the worker is said below instead.
|
||||
browser = wants_browser(args.task_dir)
|
||||
assignments = {
|
||||
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn),
|
||||
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
|
||||
"MODEL": normalize_model(harness, model),
|
||||
"EFFORT_KWARG": harness.effort_kwarg,
|
||||
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
|
||||
@@ -398,8 +420,17 @@ def main(argv: list[str] | None = None) -> int:
|
||||
warn(
|
||||
f"{harness.label} · model={assignments['MODEL']} · "
|
||||
f"{'multi-turn' if multi_turn else 'single-turn'} · "
|
||||
f"{'browser · ' if browser else ''}"
|
||||
f"agent={assignments['AGENT_IMPORT_PATH']}"
|
||||
)
|
||||
if browser and not harness.agent_import_path_browser:
|
||||
# Not a failure: the image still gets Playwright and the agent is still told about
|
||||
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
|
||||
# letting someone infer from a log line that the opt-in was ignored entirely.
|
||||
warn(
|
||||
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
|
||||
f"The browser and its disclosure are unaffected."
|
||||
)
|
||||
if harness.flaky_hangs:
|
||||
warn(
|
||||
f"{harness.label} is known to hang with no client-side timeout on a small "
|
||||
|
||||
@@ -113,11 +113,24 @@ harness_install_launchers() {
|
||||
mkdir -p "$bin"
|
||||
# Read at launcher run time so the note stays a file, not a baked-in copy.
|
||||
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
|
||||
local browser_note_src="${note_src%.md}_browser.md"
|
||||
local read_note_src="${note_src%.md}_read.md"
|
||||
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
|
||||
|
||||
local id cli launch
|
||||
local id cli launch switchable
|
||||
while IFS=$'\t' read -r id cli launch; do
|
||||
[ -n "$cli" ] && [ -n "$launch" ] || continue
|
||||
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
|
||||
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
|
||||
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
|
||||
# so a browser task needs nothing added and the flag has nothing to switch.
|
||||
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
|
||||
# which every launch line references, and every harness would look switchable.
|
||||
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
|
||||
switchable=1
|
||||
else
|
||||
switchable=0
|
||||
fi
|
||||
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
|
||||
#!/bin/bash
|
||||
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
|
||||
@@ -127,7 +140,38 @@ if [ -f "$note_src" ]; then
|
||||
else
|
||||
RACCOON_TOOLSET_NOTE=""
|
||||
fi
|
||||
export RACCOON_TOOLSET_NOTE
|
||||
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
|
||||
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
|
||||
# here. Per invocation, not per container — authoring a browser task shouldn't need a
|
||||
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
|
||||
# still mirrors an ordinary trial.
|
||||
#
|
||||
# The correction must be appended AFTER the base note, which says there is no Read tool.
|
||||
RACCOON_TOOLS="Bash"
|
||||
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
|
||||
RACCOON_TOOLS="Bash,Read"
|
||||
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
|
||||
|
||||
\$(cat "$read_note_src")"
|
||||
fi
|
||||
export RACCOON_TOOLS
|
||||
# Only mention the browser on an image that actually has one — most don't. Probed at
|
||||
# launch, not baked in, so the same launcher is correct in whichever container it runs.
|
||||
#
|
||||
# Exported two ways because the harnesses take extra instructions differently: claude
|
||||
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
|
||||
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
|
||||
# (it has no str_replace_editor), so the browser part is exported on its own too.
|
||||
RACCOON_BROWSER_NOTE=""
|
||||
RACCOON_BROWSER_FLAGS=()
|
||||
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
|
||||
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
|
||||
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
|
||||
|
||||
\${RACCOON_BROWSER_NOTE}"
|
||||
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
|
||||
fi
|
||||
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
|
||||
export RACCOON_HARNESS="$id"
|
||||
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
|
||||
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
|
||||
@@ -143,11 +187,33 @@ LAUNCHER
|
||||
|
||||
# Alias lines for ~/.bashrc.
|
||||
harness_alias_lines() {
|
||||
local id cli launch
|
||||
local id cli launch switchable
|
||||
local browser_clis=""
|
||||
while IFS=$'\t' read -r id cli launch; do
|
||||
[ -n "$cli" ] && [ -n "$launch" ] || continue
|
||||
echo "alias $cli=\"raccoon-explore-$cli\""
|
||||
# Same derivation as the launcher: only a harness whose launch line takes
|
||||
# $RACCOON_TOOLS has a toolset the flag can change.
|
||||
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
|
||||
browser_clis="${browser_clis:+$browser_clis }$cli"
|
||||
fi
|
||||
done < <(_harness_query --explore-launchers 2>/dev/null || true)
|
||||
|
||||
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
|
||||
# alternate screen buffer, so anything printed just before exec is hidden for the whole
|
||||
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
|
||||
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
|
||||
#
|
||||
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
|
||||
# and in one without.
|
||||
[ -n "$browser_clis" ] || return 0
|
||||
local first="${browser_clis%% *}"
|
||||
cat <<HINT
|
||||
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
|
||||
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
|
||||
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
|
||||
fi
|
||||
HINT
|
||||
}
|
||||
|
||||
# Write each harness's config file from the registry, replacing whatever was there.
|
||||
|
||||
@@ -557,6 +557,9 @@ repo = "${repoName}"
|
||||
commit = "${commitShort}"
|
||||
snapshot = "${basename(snapshotDir)}"
|
||||
session_uuid = "${sessionUuid}"
|
||||
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
|
||||
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
|
||||
browser = false
|
||||
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
|
||||
|
||||
[verifier]
|
||||
|
||||
@@ -22,6 +22,7 @@ import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import atif_session
|
||||
import browser_note
|
||||
from harbor.agents.installed.claude_code import ClaudeCode
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
|
||||
@@ -101,7 +102,7 @@ _AGENT_CLI_NOTE_FALLBACK = (
|
||||
)
|
||||
|
||||
|
||||
def _toolset_note() -> str:
|
||||
def _toolset_note(with_browser: bool = False, with_read: bool = False) -> str:
|
||||
"""The toolset note appended to Claude Code's stock ``--print`` system prompt
|
||||
(via ``--append-system-prompt``) for the canonical reduced toolset.
|
||||
|
||||
@@ -118,9 +119,23 @@ def _toolset_note() -> str:
|
||||
file is missing."""
|
||||
path = Path(__file__).resolve().parent / "toolset_note.md"
|
||||
try:
|
||||
return path.read_text(encoding="utf-8").strip()
|
||||
note = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
return _AGENT_CLI_NOTE_FALLBACK
|
||||
note = _AGENT_CLI_NOTE_FALLBACK
|
||||
# Order matters: the Read correction must come AFTER the base note, because it supersedes
|
||||
# that note's "there are no Read/Grep/Glob tools" line. Shipping the base note alone to a
|
||||
# Read-enabled agent would be a false statement about its own toolset.
|
||||
if with_read:
|
||||
read_note = Path(__file__).resolve().parent / "toolset_note_read.md"
|
||||
try:
|
||||
note = f"{note}\n\n{read_note.read_text(encoding='utf-8').strip()}"
|
||||
except OSError:
|
||||
_log.warning("toolset_note_read.md missing; Read correction omitted")
|
||||
if with_browser:
|
||||
extra = browser_note.browser_note()
|
||||
if extra:
|
||||
note = f"{note}\n\n{extra}"
|
||||
return note
|
||||
|
||||
|
||||
class PreinstalledClaudeCode(ClaudeCode):
|
||||
@@ -138,6 +153,10 @@ class PreinstalledClaudeCode(ClaudeCode):
|
||||
in the trial config's ``agent.import_path``.)
|
||||
"""
|
||||
|
||||
# Set by _probe_browser() during install(); read by build_cli_flags(). Declared
|
||||
# here so the full-toolset subclass (which skips the probe) still has a value.
|
||||
_has_browser = False
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "claude-code-reduced-toolset"
|
||||
@@ -176,6 +195,14 @@ class PreinstalledClaudeCode(ClaudeCode):
|
||||
editor CLI) and reuses ``_ensure_claude_binary`` directly."""
|
||||
await self._stage_agent_cli(environment)
|
||||
await self._ensure_claude_binary(environment)
|
||||
await self._probe_browser(environment)
|
||||
|
||||
async def _probe_browser(self, environment) -> None:
|
||||
"""Record whether this image ships the Playwright `pw` wrapper, so the toolset
|
||||
note mentions the browser only on images that have one. Runs during install(),
|
||||
which harbor calls before build_cli_flags() reads the result. The probe and the
|
||||
note text are shared with the codex adapter via browser_note.py."""
|
||||
self._has_browser = await browser_note.probe_browser(environment)
|
||||
|
||||
async def _ensure_claude_binary(self, environment) -> None:
|
||||
"""Reuse the claude binary already baked into the task image instead
|
||||
@@ -243,7 +270,8 @@ class PreinstalledClaudeCode(ClaudeCode):
|
||||
subclass overrides this back to stock ``ClaudeCode.build_cli_flags``.
|
||||
"""
|
||||
flags = super().build_cli_flags()
|
||||
extra = f"--tools Bash --append-system-prompt {shlex.quote(_toolset_note())}"
|
||||
note = _toolset_note(with_browser=self._has_browser)
|
||||
extra = f"--tools Bash --append-system-prompt {shlex.quote(note)}"
|
||||
return f"{flags} {extra}" if flags else extra
|
||||
|
||||
async def _claude_format_session_path(self, environment, env, session_uuid: str) -> str:
|
||||
@@ -845,6 +873,47 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
|
||||
# once every snapshot session.jsonl is re-recorded in the reduced format.
|
||||
|
||||
|
||||
class _BrowserToolsetMixin:
|
||||
"""The canonical reduced toolset PLUS the ``Read`` built-in, for tasks that opt into a
|
||||
browser: a screenshot is only useful to an agent that can look at it, and ``Read`` is what
|
||||
turns a PNG on disk into an image the model actually sees.
|
||||
|
||||
This is a SEPARATE AGENT, not a flag, per the rule that the toolset is chosen by which class
|
||||
harbor runs and the class name records it — so a benchmark row can never silently compare an
|
||||
agent that could see against one that couldn't.
|
||||
|
||||
Two things to be clear-eyed about:
|
||||
* ``Read`` is not image-only. It also reads text files, PDFs and notebooks, so these tasks
|
||||
get back a file-reading built-in the reduced toolset deliberately removes. There is no
|
||||
narrower built-in; an image-only MCP tool was rejected because the canonical agent avoids
|
||||
MCP (see the async-MCP startup race note at the top of this file).
|
||||
* Tasks on this agent are not comparable with tasks on the canonical one. That is the point
|
||||
of the distinct name.
|
||||
"""
|
||||
|
||||
def build_cli_flags(self) -> str:
|
||||
flags = ClaudeCode.build_cli_flags(self)
|
||||
note = _toolset_note(with_browser=self._has_browser, with_read=True)
|
||||
extra = f"--tools Bash,Read --append-system-prompt {shlex.quote(note)}"
|
||||
return f"{flags} {extra}" if flags else extra
|
||||
|
||||
|
||||
class BrowserPreinstalledClaudeCode(_BrowserToolsetMixin, PreinstalledClaudeCode):
|
||||
"""Reduced toolset + Read, manual (non-snapshot) tasks."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "claude-code-reduced-toolset-browser"
|
||||
|
||||
|
||||
class BrowserSnapshotClaudeCode(_BrowserToolsetMixin, SnapshotClaudeCode):
|
||||
"""Reduced toolset + Read, snapshot tasks."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "snapshot-claude-code-reduced-toolset-browser"
|
||||
|
||||
|
||||
class _FullToolsetMixin:
|
||||
"""Override the canonical reduced toolset back to Claude Code's stock full
|
||||
built-in toolset: no str_replace_editor CLI to stage, and no --tools / note.
|
||||
|
||||
@@ -926,6 +926,33 @@ if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1
|
||||
execSync(tarCmd, { stdio: 'pipe' });
|
||||
};
|
||||
|
||||
const taskDir = join('harbor-tasks', slug);
|
||||
|
||||
// tar records each file's mode as-is, and this container runs as root — so a
|
||||
// write-only file (Claude Code writes subagent session records --w-------) is
|
||||
// archived, not refused, and every later extraction of it is unreadable.
|
||||
// Normalize before packing; the catch below still repairs what only tar can see.
|
||||
try {
|
||||
const pre = normalizeTreePermissions(taskDir);
|
||||
if (didRepair(pre)) {
|
||||
log.info(
|
||||
{ ownerFixed: pre.ownerFixed.length, modeFixed: pre.modeFixed.length },
|
||||
'Normalized workspace permissions before packaging'
|
||||
);
|
||||
}
|
||||
if (pre.failures.length > 0) {
|
||||
log.warn(
|
||||
{ count: pre.failures.length, paths: pre.failures.slice(0, 5).map((f) => f.path) },
|
||||
`Could not normalize some paths. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}`
|
||||
);
|
||||
}
|
||||
} catch (repairErr) {
|
||||
log.warn(
|
||||
{ err: repairErr },
|
||||
`Permission normalization failed — packaging anyway. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}`
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
runTar();
|
||||
} catch (err) {
|
||||
@@ -952,7 +979,6 @@ if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1
|
||||
}
|
||||
}
|
||||
|
||||
const taskDir = join('harbor-tasks', slug);
|
||||
// Guarded so a failed repair can't mask the real packaging error.
|
||||
let perms;
|
||||
try {
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
## Browser
|
||||
|
||||
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
|
||||
`require("playwright")` resolvable (CommonJS — `import` will not find it).
|
||||
@@ -0,0 +1,7 @@
|
||||
## Correction to the toolset above: you also have `Read`
|
||||
|
||||
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
|
||||
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
|
||||
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
|
||||
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
|
||||
edit files with `str_replace_editor`.
|
||||
Reference in New Issue
Block a user