ren worker folder adding orig, mv new one into root
This commit is contained in:
436
worker-toolkit-potion-polyglot-orig/scripts/atif_session.py
Normal file
436
worker-toolkit-potion-polyglot-orig/scripts/atif_session.py
Normal file
@@ -0,0 +1,436 @@
|
||||
"""Convert seed conversations between harness-native session formats, via ATIF.
|
||||
|
||||
Parsers turn a native session into ATIF; renderers turn ATIF back into a native
|
||||
session. Adding a harness is one parser plus one renderer.
|
||||
|
||||
Renderers flatten tool calls to narration (`[ran Bash: {...}]` / `[result: ...]`)
|
||||
rather than rebuilding native tool-call records. Every renderer must flatten
|
||||
identically — see flatten_steps.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
ATIF_SCHEMA_VERSION = "ATIF-v1.7"
|
||||
|
||||
# Codex rollout record types, used to tell the formats apart.
|
||||
_CODEX_ROLLOUT_TYPES = frozenset(
|
||||
{"session_meta", "response_item", "event_msg", "turn_context", "compacted"}
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Format detection
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def detect_format(text: str) -> str | None:
|
||||
"""Return 'atif', 'claude', 'codex', or None for an unrecognised/empty blob."""
|
||||
stripped = text.strip()
|
||||
if not stripped:
|
||||
return None
|
||||
|
||||
# ATIF is a single JSON object, not JSONL.
|
||||
if stripped.startswith("{") and '"steps"' in stripped:
|
||||
try:
|
||||
doc = json.loads(stripped)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
doc = None
|
||||
if isinstance(doc, dict) and isinstance(doc.get("steps"), list):
|
||||
return "atif"
|
||||
|
||||
for raw in stripped.splitlines():
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
if not isinstance(rec, dict):
|
||||
continue
|
||||
if rec.get("type") in _CODEX_ROLLOUT_TYPES and "message" not in rec:
|
||||
return "codex"
|
||||
if rec.get("type") in ("user", "assistant") or "message" in rec:
|
||||
return "claude"
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ATIF construction helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _trajectory(steps: list[dict], *, session_id: str | None = None) -> dict:
|
||||
return {
|
||||
"schema_version": ATIF_SCHEMA_VERSION,
|
||||
"session_id": session_id,
|
||||
"agent": {"name": "unknown"},
|
||||
"steps": steps,
|
||||
}
|
||||
|
||||
|
||||
def _step(
|
||||
step_id: int,
|
||||
source: str,
|
||||
*,
|
||||
message: str = "",
|
||||
reasoning: str | None = None,
|
||||
tool_calls: list[dict] | None = None,
|
||||
observations: list[dict] | None = None,
|
||||
timestamp: str | None = None,
|
||||
) -> dict:
|
||||
step: dict[str, Any] = {
|
||||
"step_id": step_id,
|
||||
"source": source,
|
||||
"message": message,
|
||||
"is_copied_context": True,
|
||||
}
|
||||
if timestamp:
|
||||
step["timestamp"] = timestamp
|
||||
if reasoning:
|
||||
step["reasoning_content"] = reasoning
|
||||
if tool_calls:
|
||||
step["tool_calls"] = tool_calls
|
||||
if observations:
|
||||
step["observation"] = {"results": observations}
|
||||
return step
|
||||
|
||||
|
||||
def _content_text(content: Any) -> str:
|
||||
"""Text of an ATIF message or a ContentPart list."""
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if isinstance(content, list):
|
||||
return "".join(
|
||||
part.get("text") or "" for part in content if isinstance(part, dict)
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Parsers: native -> ATIF
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def claude_session_to_atif(jsonl_text: str) -> dict:
|
||||
"""Parse a Claude Code session.jsonl into ATIF steps.
|
||||
|
||||
Claude records tool results on `user` records; they become observations.
|
||||
"""
|
||||
steps: list[dict] = []
|
||||
session_id: str | None = None
|
||||
|
||||
for raw in jsonl_text.splitlines():
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
if not isinstance(rec, dict):
|
||||
continue
|
||||
session_id = session_id or rec.get("sessionId")
|
||||
|
||||
msg = rec.get("message") or {}
|
||||
role = msg.get("role") or rec.get("type")
|
||||
if role not in ("user", "assistant"):
|
||||
continue
|
||||
content = msg.get("content")
|
||||
if content is None:
|
||||
continue
|
||||
|
||||
source = "user" if role == "user" else "agent"
|
||||
if isinstance(content, str):
|
||||
steps.append(
|
||||
_step(
|
||||
len(steps) + 1, source, message=content, timestamp=rec.get("timestamp")
|
||||
)
|
||||
)
|
||||
continue
|
||||
|
||||
text_parts: list[str] = []
|
||||
reasoning_parts: list[str] = []
|
||||
tool_calls: list[dict] = []
|
||||
observations: list[dict] = []
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
text_parts.append(str(block))
|
||||
continue
|
||||
btype = block.get("type")
|
||||
if btype == "text":
|
||||
text_parts.append(block.get("text") or "")
|
||||
elif btype == "thinking":
|
||||
reasoning_parts.append(block.get("thinking") or "")
|
||||
elif btype == "tool_use":
|
||||
tool_calls.append(
|
||||
{
|
||||
"tool_call_id": block.get("id") or f"call_{len(tool_calls) + 1}",
|
||||
"function_name": block.get("name") or "tool",
|
||||
"arguments": block.get("input") or {},
|
||||
}
|
||||
)
|
||||
elif btype == "tool_result":
|
||||
observations.append(
|
||||
{
|
||||
"source_call_id": block.get("tool_use_id"),
|
||||
"content": _content_text(block.get("content")),
|
||||
}
|
||||
)
|
||||
|
||||
steps.append(
|
||||
_step(
|
||||
len(steps) + 1,
|
||||
source,
|
||||
message="".join(text_parts),
|
||||
reasoning="".join(reasoning_parts) or None,
|
||||
tool_calls=tool_calls or None,
|
||||
observations=observations or None,
|
||||
timestamp=rec.get("timestamp"),
|
||||
)
|
||||
)
|
||||
|
||||
return _trajectory(steps, session_id=session_id)
|
||||
|
||||
|
||||
def codex_rollout_to_atif(jsonl_text: str) -> dict:
|
||||
"""Parse a codex rollout JSONL into ATIF steps."""
|
||||
steps: list[dict] = []
|
||||
session_id: str | None = None
|
||||
|
||||
for raw in jsonl_text.splitlines():
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
if not isinstance(rec, dict):
|
||||
continue
|
||||
|
||||
rtype = rec.get("type")
|
||||
payload = rec.get("payload") or {}
|
||||
if rtype == "session_meta":
|
||||
session_id = session_id or payload.get("id")
|
||||
continue
|
||||
if rtype != "response_item":
|
||||
continue
|
||||
|
||||
ptype = payload.get("type")
|
||||
timestamp = rec.get("timestamp")
|
||||
|
||||
if ptype == "message":
|
||||
role = payload.get("role")
|
||||
if role not in ("user", "assistant"):
|
||||
continue
|
||||
steps.append(
|
||||
_step(
|
||||
len(steps) + 1,
|
||||
"user" if role == "user" else "agent",
|
||||
message=_content_text(payload.get("content")),
|
||||
timestamp=timestamp,
|
||||
)
|
||||
)
|
||||
elif ptype in ("function_call", "local_shell_call", "custom_tool_call"):
|
||||
arguments = payload.get("arguments")
|
||||
if isinstance(arguments, str):
|
||||
try:
|
||||
arguments = json.loads(arguments)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
arguments = {"raw": arguments}
|
||||
steps.append(
|
||||
_step(
|
||||
len(steps) + 1,
|
||||
"agent",
|
||||
tool_calls=[
|
||||
{
|
||||
"tool_call_id": payload.get("call_id") or f"call_{len(steps)}",
|
||||
"function_name": payload.get("name") or "tool",
|
||||
"arguments": arguments or {},
|
||||
}
|
||||
],
|
||||
timestamp=timestamp,
|
||||
)
|
||||
)
|
||||
elif ptype in ("function_call_output", "custom_tool_call_output"):
|
||||
output = payload.get("output")
|
||||
if isinstance(output, dict):
|
||||
output = output.get("content") or json.dumps(output, ensure_ascii=False)
|
||||
steps.append(
|
||||
_step(
|
||||
len(steps) + 1,
|
||||
"agent",
|
||||
observations=[
|
||||
{
|
||||
"source_call_id": payload.get("call_id"),
|
||||
"content": output if isinstance(output, str) else "",
|
||||
}
|
||||
],
|
||||
timestamp=timestamp,
|
||||
)
|
||||
)
|
||||
elif ptype == "reasoning":
|
||||
summary = payload.get("summary")
|
||||
text = ""
|
||||
if isinstance(summary, list):
|
||||
text = "".join(
|
||||
s.get("text") or "" for s in summary if isinstance(s, dict)
|
||||
)
|
||||
if text:
|
||||
steps.append(_step(len(steps) + 1, "agent", reasoning=text, timestamp=timestamp))
|
||||
|
||||
return _trajectory(steps, session_id=session_id)
|
||||
|
||||
|
||||
def to_atif(text: str) -> dict:
|
||||
"""Parse whichever native format `text` is into ATIF."""
|
||||
fmt = detect_format(text)
|
||||
if fmt == "atif":
|
||||
return json.loads(text)
|
||||
if fmt == "claude":
|
||||
return claude_session_to_atif(text)
|
||||
if fmt == "codex":
|
||||
return codex_rollout_to_atif(text)
|
||||
raise ValueError("unrecognised session format (not ATIF, Claude JSONL, or codex rollout)")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Flattening — shared by every renderer so the loss stays symmetric
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def flatten_steps(atif: dict) -> list[tuple[str, str]]:
|
||||
"""ATIF steps -> ordered (role, text) pairs, where role is 'user' or 'agent'.
|
||||
|
||||
Tool calls and observations become agent narration; `reasoning_content` is dropped.
|
||||
"""
|
||||
out: list[tuple[str, str]] = []
|
||||
for step in atif.get("steps") or []:
|
||||
if not isinstance(step, dict):
|
||||
continue
|
||||
role = "user" if step.get("source") == "user" else "agent"
|
||||
|
||||
text = _content_text(step.get("message"))
|
||||
if text:
|
||||
out.append((role, text))
|
||||
|
||||
for call in step.get("tool_calls") or []:
|
||||
if not isinstance(call, dict):
|
||||
continue
|
||||
args = json.dumps(call.get("arguments") or {}, ensure_ascii=False)
|
||||
out.append(("agent", f"[ran {call.get('function_name') or 'tool'}: {args}]"))
|
||||
|
||||
observation = step.get("observation") or {}
|
||||
for result in observation.get("results") or []:
|
||||
if not isinstance(result, dict):
|
||||
continue
|
||||
out.append(("agent", f"[result: {_content_text(result.get('content'))}]"))
|
||||
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Renderers: ATIF -> native
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def atif_to_codex_rollout(
|
||||
atif: dict,
|
||||
iso_ts: str,
|
||||
*,
|
||||
session_meta: dict,
|
||||
max_total: int | None = None,
|
||||
) -> list[str]:
|
||||
"""Render ATIF as codex rollout JSONL that `codex exec resume` can continue.
|
||||
|
||||
`session_meta` is supplied by the caller so this module reads no files.
|
||||
"""
|
||||
lines = [json.dumps(session_meta)]
|
||||
budget = float("inf") if max_total is None else max_total
|
||||
|
||||
for role, text in flatten_steps(atif):
|
||||
text = (text or "").strip()
|
||||
if not text or budget <= 0:
|
||||
continue
|
||||
text = text[: int(min(budget, len(text)))]
|
||||
ctype = "input_text" if role == "user" else "output_text"
|
||||
lines.append(
|
||||
json.dumps(
|
||||
{
|
||||
"timestamp": iso_ts,
|
||||
"type": "response_item",
|
||||
"payload": {
|
||||
"type": "message",
|
||||
"role": "user" if role == "user" else "assistant",
|
||||
"content": [{"type": ctype, "text": text}],
|
||||
},
|
||||
}
|
||||
)
|
||||
)
|
||||
budget -= len(text)
|
||||
|
||||
return lines
|
||||
|
||||
|
||||
def atif_to_claude_session(
|
||||
atif: dict,
|
||||
*,
|
||||
session_id: str,
|
||||
cwd: str = "/workspace",
|
||||
git_branch: str = "main",
|
||||
version: str = "2.1.87",
|
||||
iso_ts: str,
|
||||
max_total: int | None = None,
|
||||
) -> list[str]:
|
||||
"""Render ATIF as Claude Code session.jsonl that `claude --resume` can continue.
|
||||
|
||||
Records are chained by parentUuid: Claude resumes by walking that chain, not by
|
||||
file order.
|
||||
"""
|
||||
lines: list[str] = []
|
||||
parent_uuid: str | None = None
|
||||
budget = float("inf") if max_total is None else max_total
|
||||
|
||||
for index, (role, text) in enumerate(flatten_steps(atif), start=1):
|
||||
text = (text or "").strip()
|
||||
if not text or budget <= 0:
|
||||
continue
|
||||
text = text[: int(min(budget, len(text)))]
|
||||
uuid = _deterministic_uuid(session_id, index)
|
||||
claude_role = "user" if role == "user" else "assistant"
|
||||
record: dict[str, Any] = {
|
||||
"parentUuid": parent_uuid,
|
||||
"isSidechain": False,
|
||||
"userType": "external",
|
||||
"cwd": cwd,
|
||||
"sessionId": session_id,
|
||||
"version": version,
|
||||
"gitBranch": git_branch,
|
||||
"type": claude_role,
|
||||
"uuid": uuid,
|
||||
"timestamp": iso_ts,
|
||||
}
|
||||
if claude_role == "user":
|
||||
record["message"] = {"role": "user", "content": text}
|
||||
else:
|
||||
record["message"] = {
|
||||
"role": "assistant",
|
||||
"content": [{"type": "text", "text": text}],
|
||||
"stop_reason": "end_turn",
|
||||
}
|
||||
lines.append(json.dumps(record))
|
||||
parent_uuid = uuid
|
||||
budget -= len(text)
|
||||
|
||||
return lines
|
||||
|
||||
|
||||
def _deterministic_uuid(session_id: str, index: int) -> str:
|
||||
"""A stable uuid5 per (session, position), so re-rendering is byte-identical."""
|
||||
import uuid as _uuid
|
||||
|
||||
return str(_uuid.uuid5(_uuid.NAMESPACE_URL, f"raccoon-seed/{session_id}/{index}"))
|
||||
53
worker-toolkit-potion-polyglot-orig/scripts/browser_note.py
Normal file
53
worker-toolkit-potion-polyglot-orig/scripts/browser_note.py
Normal file
@@ -0,0 +1,53 @@
|
||||
"""Shared browser-capability disclosure for the agent harnesses.
|
||||
|
||||
Only images for browser-facing repos ship Playwright, so the note is conditional on probing
|
||||
the sandbox for the `pw` wrapper rather than on anything about the task. Probing keeps the
|
||||
claim true by construction: telling an agent it has a browser it does not have sends it after
|
||||
a missing binary. To check an image yourself: `command -v pw`.
|
||||
|
||||
Both harnesses disclose the same text through their own mechanism:
|
||||
- Claude Code: appended to --append-system-prompt (scripts/snapshot_agent.py)
|
||||
- codex: -c developer_instructions=... (scripts/codex_agent.py), which prepends a
|
||||
developer message and LEAVES codex's base instructions intact. Verified with
|
||||
`codex debug prompt-input`. Do not switch to model_instructions_file — that
|
||||
REPLACES the base instructions.
|
||||
|
||||
This module exists so the probe and the text live in one place; a copy in each adapter would
|
||||
drift and the drift would be invisible (both would still run, just disclosing differently).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
_log = logging.getLogger(__name__)
|
||||
|
||||
_NOTE_FILE = Path(__file__).resolve().parent / "toolset_note_browser.md"
|
||||
_PROBE = "command -v pw >/dev/null 2>&1 && echo yes || echo no"
|
||||
|
||||
|
||||
def browser_note() -> str:
|
||||
"""The disclosure text, or "" if the note file is missing (never fatal)."""
|
||||
try:
|
||||
return _NOTE_FILE.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_log.warning("%s missing; browser note omitted", _NOTE_FILE.name)
|
||||
return ""
|
||||
|
||||
|
||||
async def probe_browser(environment) -> bool:
|
||||
"""True when this image ships the `pw` wrapper. Best-effort: a failed probe means no
|
||||
note, never a failed run."""
|
||||
try:
|
||||
result = await environment.exec(command=_PROBE, timeout_sec=30)
|
||||
except Exception as exc:
|
||||
_log.warning("browser probe failed (%s); omitting the browser note", exc)
|
||||
return False
|
||||
# Exact tail match, not a substring: several harbor environments exec through a LOGIN
|
||||
# shell, whose profile scripts can print to stdout. A banner containing "yes" would
|
||||
# otherwise claim a browser that isn't there — the precise failure this module exists
|
||||
# to prevent.
|
||||
found = (getattr(result, "stdout", "") or "").strip().endswith("yes")
|
||||
_log.info("browser probe: pw %s", "present" if found else "absent")
|
||||
return found
|
||||
325
worker-toolkit-potion-polyglot-orig/scripts/build-workspace.sh
Executable file
325
worker-toolkit-potion-polyglot-orig/scripts/build-workspace.sh
Executable file
@@ -0,0 +1,325 @@
|
||||
#!/bin/bash
|
||||
# Build a task's workspace from the local repo.
|
||||
#
|
||||
# Usage: scripts/build-workspace.sh <task-slug> [commit]
|
||||
# Example: scripts/build-workspace.sh my-cool-task 3af4366a6
|
||||
#
|
||||
# If commit is omitted, reads it from the task's task.toml.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
TOOLKIT_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
REPO_DIR="$TOOLKIT_ROOT/repo"
|
||||
|
||||
TASK_SLUG="$1"
|
||||
TASK_DIR="$TOOLKIT_ROOT/harbor-tasks/$TASK_SLUG"
|
||||
|
||||
if [ ! -d "$TASK_DIR" ]; then
|
||||
echo "Error: task directory not found at $TASK_DIR" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The member this task targets, per task.toml ([metadata].repo). Used to resolve both
|
||||
# the source repo (polyglot) and the member's deterministic checks (below).
|
||||
# `|| true` is load-bearing: a task.toml with no `repo =` line is perfectly valid
|
||||
# (single-repo tasks don't need one), but under `set -o pipefail` grep's exit 1
|
||||
# propagates out of the pipeline and `set -e` would kill the script here.
|
||||
MEMBER=""
|
||||
if [ -f "$TASK_DIR/task.toml" ]; then
|
||||
MEMBER=$(grep -E '^repo[[:space:]]*=' "$TASK_DIR/task.toml" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)
|
||||
fi
|
||||
|
||||
# Single-repo toolkits keep the repo at $ROOT/repo; a polyglot toolkit keeps each member
|
||||
# at $ROOT/repos/<member>. If the single-repo path is absent, use the member from task.toml
|
||||
# so a graded task builds against the right member repo.
|
||||
if [ ! -d "$REPO_DIR/.git" ] && [ -n "$MEMBER" ]; then
|
||||
if [ -d "$TOOLKIT_ROOT/repos/$MEMBER/.git" ]; then
|
||||
REPO_DIR="$TOOLKIT_ROOT/repos/$MEMBER"
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ ! -d "$REPO_DIR/.git" ]; then
|
||||
echo "Error: repo not found at $REPO_DIR" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Get commit from arg or task.toml
|
||||
if [ -n "${2:-}" ]; then
|
||||
COMMIT="$2"
|
||||
else
|
||||
# `|| true` for the same reason as MEMBER above: without it, pipefail turns a
|
||||
# task.toml with no `commit` line into a bare `set -e` abort, and the explicit
|
||||
# error below never gets a chance to print.
|
||||
COMMIT=$(grep 'commit' "$TASK_DIR/task.toml" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)
|
||||
if [ -z "$COMMIT" ]; then
|
||||
echo "Error: no commit specified and could not read from task.toml" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
WORKSPACE="$TASK_DIR/environment/workspace"
|
||||
|
||||
echo "Building workspace for $TASK_SLUG"
|
||||
echo " Commit: $COMMIT"
|
||||
|
||||
# `browser = true` in task.toml gives the trial Playwright + Chromium. The build has no way
|
||||
# to read task.toml — a Dockerfile can only see its build context — so the answer is written
|
||||
# here as a file the Dockerfile COPYs.
|
||||
#
|
||||
# ALWAYS write it, including the "0" case: the COPY is unconditional, and a missing source
|
||||
# fails the build. Accepts `true` and `"true"`, since the quoted form is a plausible hand-edit
|
||||
# and rejecting it would silently give a task no browser after its author asked for one.
|
||||
BROWSER_OPTIN=0
|
||||
if [ -f "$TASK_DIR/task.toml" ] &&
|
||||
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
|
||||
BROWSER_OPTIN=1
|
||||
fi
|
||||
mkdir -p "$TASK_DIR/environment"
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
|
||||
|
||||
# --- DNS jail script staging -------------------------------------------------
|
||||
# The Dockerfile COPYs this directory, so it must always exist (same rule as the marker
|
||||
# above: a missing COPY source fails the build). The script itself is optional -- without it
|
||||
# the image installs no resolver and trials simply run with normal network access.
|
||||
mkdir -p "$TASK_DIR/environment/dns-jail"
|
||||
if [ -f "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" ]; then
|
||||
cp "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" \
|
||||
"$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
|
||||
fi
|
||||
# Not every member's image ships a browser, and the Explore container has one either way — so
|
||||
# a task can ask for a browser it will not get. Say so here rather than let it pass silently.
|
||||
if [ "$BROWSER_OPTIN" = "1" ]; then
|
||||
if grep -q "COPY browser-optin" "$TASK_DIR/environment/Dockerfile" 2>/dev/null; then
|
||||
echo " Browser: Playwright + Chromium (browser = true)"
|
||||
else
|
||||
echo " WARNING: browser = true, but this task's Dockerfile has no browser. The agent" >&2
|
||||
echo " will get the Read tool and no Chromium. Either drop the flag, or use a" >&2
|
||||
echo " member whose image ships one:" >&2
|
||||
echo " grep -l 'COPY browser-optin' task-shared/Dockerfile.*" >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# Resolve commit
|
||||
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
|
||||
echo " Resolved SHA: $RESOLVED_SHA"
|
||||
|
||||
# --- Member-specific setup ---------------------------------------------------
|
||||
#
|
||||
# A polyglot _task-scaffold can't know which member a task targets, so anything
|
||||
# member-specific is resolved here instead of left to the author to remember. This is
|
||||
# the one step every task runs on both the manual and snapshot paths, and task.toml
|
||||
# already tells us the member. Both actions below are idempotent and never clobber
|
||||
# authored content, so re-running is always safe.
|
||||
SHARED_DIR="$TOOLKIT_ROOT/task-shared"
|
||||
MEMBER_LC=$(echo "${MEMBER:-}" | tr '[:upper:]' '[:lower:]')
|
||||
|
||||
# 1. Base image. Replace the placeholder Dockerfile with the member's real base. Guarded
|
||||
# on the placeholder marker so an authored Dockerfile is never touched — snapshot tasks
|
||||
# append session staging to theirs, and any task may be customized by hand. The marker
|
||||
# must match POLYGLOT_SCAFFOLD_DOCKERFILE in package-worker-toolkit.ts; a packaging test
|
||||
# asserts the two agree so this can't silently stop matching.
|
||||
TASK_DOCKERFILE="$TASK_DIR/environment/Dockerfile"
|
||||
if [ -f "$TASK_DOCKERFILE" ] && grep -q 'POLYGLOT TOOLKIT' "$TASK_DOCKERFILE"; then
|
||||
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/Dockerfile.$MEMBER_LC" ]; then
|
||||
cp "$SHARED_DIR/Dockerfile.$MEMBER_LC" "$TASK_DOCKERFILE"
|
||||
echo " Set base image: environment/Dockerfile (from Dockerfile.$MEMBER_LC)"
|
||||
else
|
||||
echo " WARN: environment/Dockerfile is still the scaffold placeholder and no" >&2
|
||||
echo " task-shared/Dockerfile.${MEMBER_LC:-<member>} exists to replace it with." >&2
|
||||
echo " Set [metadata].repo in task.toml to your member, then re-run this script." >&2
|
||||
echo " Members: $(cd "$SHARED_DIR" 2>/dev/null && ls Dockerfile.* 2>/dev/null | sed 's/Dockerfile\.//' | tr '\n' ' ')" >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# 2. Deterministic checks (tests/typecheck/lint). tests/test.sh sources this file and
|
||||
# hands its output to the grader as evidence for the CORRECTNESS score, so a task without
|
||||
# it gets a correctness score judged from the code alone — no test signal behind it. The
|
||||
# absent-only guard leaves an existing file untouched (a single-repo scaffold ships one).
|
||||
if [ ! -f "$TASK_DIR/tests/test-commands.sh" ]; then
|
||||
CHECKS_SRC=""
|
||||
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/test-commands.$MEMBER_LC.sh" ]; then
|
||||
CHECKS_SRC="$SHARED_DIR/test-commands.$MEMBER_LC.sh"
|
||||
elif [ -f "$SHARED_DIR/test-commands.sh" ]; then
|
||||
CHECKS_SRC="$SHARED_DIR/test-commands.sh"
|
||||
fi
|
||||
if [ -n "$CHECKS_SRC" ]; then
|
||||
mkdir -p "$TASK_DIR/tests"
|
||||
cp "$CHECKS_SRC" "$TASK_DIR/tests/test-commands.sh"
|
||||
chmod +x "$TASK_DIR/tests/test-commands.sh"
|
||||
echo " Staged deterministic checks: tests/test-commands.sh (from $(basename "$CHECKS_SRC"))"
|
||||
else
|
||||
# Say it out loud. Absence is legitimate for members with no runnable checks, but
|
||||
# silence is indistinguishable from a mistake — and it changes how the correctness
|
||||
# score is arrived at, so the author should know either way.
|
||||
echo " NOTE: no deterministic checks available for ${MEMBER:-this repo} — the grader will"
|
||||
echo " score correctness from the code alone, with no test/typecheck/lint signal."
|
||||
fi
|
||||
fi
|
||||
|
||||
# Clean and recreate
|
||||
rm -rf "$WORKSPACE"
|
||||
mkdir -p "$WORKSPACE"
|
||||
|
||||
# Export repo at target commit (no git history).
|
||||
# --no-same-owner: `git archive` stamps every entry as uid/gid 0, so GNU tar
|
||||
# running as (container) root tries to chown files back to 0/0. On nested /
|
||||
# rootless / Sysbox runtimes the container "root" is a userns-mapped uid with no
|
||||
# CAP_CHOWN, so that chown fails with EPERM. --no-same-owner skips the restore
|
||||
# (files are owned by the extracting user) — a no-op for real root and for
|
||||
# non-root extraction, and the fix for the mapped-root case.
|
||||
git -C "$REPO_DIR" archive "$RESOLVED_SHA" | tar -x --no-same-owner -C "$WORKSPACE"
|
||||
|
||||
# Apply workspace patch if one exists
|
||||
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
|
||||
if [ -f "$PATCH_FILE" ]; then
|
||||
echo " Applying workspace.patch..."
|
||||
cd "$WORKSPACE"
|
||||
# gc.auto=0 / maintenance.auto=false / gc.autoDetach=false prevent git
|
||||
# from launching background processes (gc, commit-graph, fsmonitor) that
|
||||
# can write into .git/objects after the foreground command returns. If
|
||||
# such a write races with the `rm -rf .git` below, rmdir trips on
|
||||
# "Directory not empty" and the build fails non-deterministically.
|
||||
GIT_FLAGS=(-c gc.auto=0 -c gc.autoDetach=false -c maintenance.auto=false)
|
||||
git "${GIT_FLAGS[@]}" init --quiet
|
||||
git "${GIT_FLAGS[@]}" add -A
|
||||
# Inject identity inline so this works on containers without a global
|
||||
# git config (e.g., native Linux Docker, fresh container images).
|
||||
# The .git directory is deleted on the next line, so these values are
|
||||
# throwaway and never reach the patch, the workspace, or the agent.
|
||||
git "${GIT_FLAGS[@]}" -c user.email=toolkit@local -c user.name=Toolkit commit -m "base" --quiet
|
||||
git "${GIT_FLAGS[@]}" apply "$PATCH_FILE"
|
||||
# Belt-and-suspenders: retry rm a few times in case anything still races.
|
||||
for _ in 1 2 3; do
|
||||
if rm -rf .git 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
# Final attempt without swallowing errors, so a genuine failure surfaces.
|
||||
if [ -d .git ]; then
|
||||
rm -rf .git
|
||||
fi
|
||||
cd "$TOOLKIT_ROOT"
|
||||
echo " Patch applied."
|
||||
fi
|
||||
|
||||
# Bundle transitive poetry sibling deps. Some polyglot Python members poetry-depend on
|
||||
# sibling repos via `ssh://git@github.com/AskZeta/<name>`, which can't resolve in a single-member
|
||||
# harbor image (no SSH key / network). Archive the transitive closure into workspace/.zeta-siblings/<name>/
|
||||
# from the toolkit's repos/zeta-<name>/ (members are packaged under their display name zeta-<name>);
|
||||
# the generated Dockerfile rewrites those git deps to
|
||||
# these local paths before `poetry install`. No-op for members without such deps.
|
||||
if [ -f "$WORKSPACE/pyproject.toml" ]; then
|
||||
SIB_DIR="$WORKSPACE/.zeta-siblings"
|
||||
queue=("$WORKSPACE/pyproject.toml")
|
||||
seen=" "
|
||||
while [ "${#queue[@]}" -gt 0 ]; do
|
||||
pp="${queue[0]}"; queue=("${queue[@]:1}")
|
||||
[ -f "$pp" ] || continue
|
||||
for name in $(grep -oE 'ssh://git@github\.com/AskZeta/[A-Za-z0-9._-]+' "$pp" 2>/dev/null | sed -E 's#.*/AskZeta/##; s#\.git$##' | sort -u); do
|
||||
case "$seen" in *" $name "*) continue ;; esac
|
||||
seen="$seen$name "
|
||||
sib="$TOOLKIT_ROOT/repos/zeta-$name"
|
||||
[ -e "$sib/.git" ] || { echo " WARN: sibling repo not found: $name" >&2; continue; }
|
||||
mkdir -p "$SIB_DIR/$name"
|
||||
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SIB_DIR/$name"
|
||||
queue+=("$SIB_DIR/$name/pyproject.toml")
|
||||
done
|
||||
done
|
||||
[ -d "$SIB_DIR" ] && echo " Bundled siblings:$(printf '%s' "${seen# }" | sed 's/ $//' | sed 's/^/ /')"
|
||||
fi
|
||||
|
||||
# Bundle Maven sibling libs. The swingbell-polyglot Java services depend on sibling shared
|
||||
# artifacts (com.swingbell*: common-repository, common-aws-service, jasper-report from
|
||||
# `reports`, jwt-encryption-decryption) at 0.0.1-SNAPSHOT — resolvable only from a local
|
||||
# reactor install, never a registry. Walk the dependency closure (a bundled provider's own
|
||||
# pom can name further siblings — `reports` needs common-repository) into
|
||||
# workspace/.sbl-siblings/<name>/ with an ORDER file in install order (commons before
|
||||
# consumers); the generated java Dockerfile `mvn install`s them into ~/.m2 before building
|
||||
# the member. No-op without a pom or refs.
|
||||
#
|
||||
# Twin forks: the book-my-minutes-* repos publish the SAME coordinates as the swingbell
|
||||
# commons (com.swingbell.common:common-repository:0.0.1-SNAPSHOT etc. — the twin naming is
|
||||
# repo-level only, invisible to Maven), so an artifactId resolves to the provider from the
|
||||
# member's own family.
|
||||
if [ -f "$WORKSPACE/pom.xml" ] && grep -q 'com\.swingbell' "$WORKSPACE/pom.xml" 2>/dev/null; then
|
||||
SBL_DIR="$WORKSPACE/.sbl-siblings"
|
||||
case "$MEMBER" in book-my-minutes-*) SBL_TWIN=book-my-minutes- ;; *) SBL_TWIN= ;; esac
|
||||
queue=("$WORKSPACE/pom.xml")
|
||||
seen=" "
|
||||
while [ "${#queue[@]}" -gt 0 ]; do
|
||||
pom="${queue[0]}"; queue=("${queue[@]:1}")
|
||||
[ -f "$pom" ] || continue
|
||||
for artifact in $(grep -oE '<artifactId>(common-repository|common-aws-service|jasper-report|jwt-encryption-decryption)</artifactId>' "$pom" 2>/dev/null | sed -E 's#</?artifactId>##g' | sort -u); do
|
||||
case "$artifact" in
|
||||
common-repository|common-aws-service) provider="$SBL_TWIN$artifact" ;;
|
||||
jasper-report) provider=reports ;;
|
||||
jwt-encryption-decryption) provider=jwt-encryption-decryption ;;
|
||||
esac
|
||||
case "$seen" in *" $provider "*) continue ;; esac
|
||||
# never bundle the member into itself: the pom's OWN <artifactId> declaration
|
||||
# matches the grep above just like a dependency would ($MEMBER is the task repo)
|
||||
[ "$provider" = "$MEMBER" ] && continue
|
||||
seen="$seen$provider "
|
||||
sib="$TOOLKIT_ROOT/repos/$provider"
|
||||
[ -e "$sib/.git" ] || { echo " WARN: maven sibling repo not found: $provider" >&2; continue; }
|
||||
mkdir -p "$SBL_DIR/$provider"
|
||||
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SBL_DIR/$provider"
|
||||
queue+=("$SBL_DIR/$provider/pom.xml")
|
||||
done
|
||||
done
|
||||
# ORDER = canonical install order (providers before their consumers), filtered to the
|
||||
# closure just bundled — discovery order is consumer-first, which is backwards for install.
|
||||
for provider in "${SBL_TWIN}common-repository" "${SBL_TWIN}common-aws-service" jwt-encryption-decryption reports; do
|
||||
case "$seen" in *" $provider "*) echo "$provider" >> "$SBL_DIR/ORDER" ;; esac
|
||||
done
|
||||
[ -f "$SBL_DIR/ORDER" ] && echo " Bundled maven siblings: $(tr '\n' ' ' < "$SBL_DIR/ORDER")"
|
||||
fi
|
||||
|
||||
# --- Reference-data corpus: mounted at /data/zeta-corpus in the trial -----------------------
|
||||
# If this toolkit ships the supplementary data corpus, it's included in every trial — staged into
|
||||
# the build context + a COPY added to the Dockerfile, so what you see while authoring (bind-mounted
|
||||
# at /data/zeta-corpus) is exactly what the trial sees. Toolkits without a corpus never include it.
|
||||
CORPUS_SRC=""
|
||||
for cand in "${ZETA_CORPUS_DIR:-}" "$TOOLKIT_ROOT/data/zeta-corpus" "/data/zeta-corpus"; do
|
||||
[ -n "$cand" ] && [ -d "$cand" ] && { CORPUS_SRC="$cand"; break; }
|
||||
done
|
||||
if [ -n "$CORPUS_SRC" ]; then
|
||||
CORPUS_STAGE="$TASK_DIR/environment/corpus"
|
||||
rm -rf "$CORPUS_STAGE"
|
||||
# hardlink-stage (cp -al ~free, same filesystem as the toolkit); full copy fallback.
|
||||
cp -al "$CORPUS_SRC/." "$CORPUS_STAGE" 2>/dev/null || cp -a "$CORPUS_SRC/." "$CORPUS_STAGE"
|
||||
DF="$TASK_DIR/environment/Dockerfile"
|
||||
# Wrapped in toolkit-managed sentinels so scripts/check-task-infra.ts can tell
|
||||
# this append apart from an author's edit — see scripts/lib/task-infra-integrity.ts.
|
||||
if [ -f "$DF" ] && ! grep -qF 'COPY corpus/ /data/zeta-corpus' "$DF"; then
|
||||
{ echo ""; echo "# >>> toolkit-managed: corpus >>>"; \
|
||||
echo "# Reference-data corpus at /data/zeta-corpus (staged by build-workspace)."; \
|
||||
echo "COPY corpus/ /data/zeta-corpus/"; \
|
||||
echo "# <<< toolkit-managed <<<"; } >> "$DF"
|
||||
fi
|
||||
if [ -f "$TASK_DIR/task.toml" ]; then
|
||||
CUR=$(grep -oE '^[[:space:]]*storage_mb[[:space:]]*=[[:space:]]*[0-9]+' "$TASK_DIR/task.toml" | grep -oE '[0-9]+' | head -1 || echo 0)
|
||||
# 10240 = the sandbox disk ceiling (a higher request is rejected downstream).
|
||||
if [ "${CUR:-0}" -lt 10240 ] && grep -qE '^[[:space:]]*storage_mb[[:space:]]*=' "$TASK_DIR/task.toml"; then
|
||||
sed -i.bak -E 's/^([[:space:]]*storage_mb[[:space:]]*=[[:space:]]*)[0-9]+/\110240/' "$TASK_DIR/task.toml"
|
||||
rm -f "$TASK_DIR/task.toml.bak"
|
||||
fi
|
||||
fi
|
||||
echo " Corpus: staged from $CORPUS_SRC -> environment/corpus + Dockerfile COPY (storage_mb>=10240)"
|
||||
fi
|
||||
|
||||
FILE_COUNT=$(find "$WORKSPACE" -type f | wc -l | tr -d ' ')
|
||||
echo " Workspace: $WORKSPACE ($FILE_COUNT files)"
|
||||
|
||||
# Toolkit-managed files. Stamp them if they aren't already (tasks copied from
|
||||
# _task-scaffold arrive stamped; this covers the ones built by snapshot-to-task), then
|
||||
# report. Advisory only — this script writes to the Dockerfile itself, so it never
|
||||
# blocks; harbor-run and submit-task do.
|
||||
CHECK_INFRA="$TOOLKIT_ROOT/scripts/check-task-infra.ts"
|
||||
if [ -f "$CHECK_INFRA" ]; then
|
||||
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" --stamp "$TASK_SLUG") || true
|
||||
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_SLUG") || true
|
||||
fi
|
||||
|
||||
echo "Done."
|
||||
138
worker-toolkit-potion-polyglot-orig/scripts/check-task-infra.ts
Normal file
138
worker-toolkit-potion-polyglot-orig/scripts/check-task-infra.ts
Normal file
@@ -0,0 +1,138 @@
|
||||
/**
|
||||
* check-task-infra.ts — report edits to toolkit-managed files.
|
||||
*
|
||||
* Called by `scripts/harbor-run` before a trial and by `scripts/submit-task.ts`
|
||||
* before packaging, so an accidental edit to the trial Dockerfile, the grader
|
||||
* orchestration, or the grader system prompt surfaces at the moment it matters
|
||||
* rather than after a submission is reviewed.
|
||||
*
|
||||
* Covers two sets: the task's own managed files (environment/Dockerfile,
|
||||
* tests/test.sh, the grader system prompts) and the toolkit's `scripts/` tree,
|
||||
* which is checked once per invocation regardless of which task was named.
|
||||
*
|
||||
* Always exits 0. Both checks are advisory — see the notes on IntegrityStatus
|
||||
* and formatIntegrityReport.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/check-task-infra.ts <task-slug-or-dir>
|
||||
* npx tsx scripts/check-task-infra.ts my-task --json
|
||||
*/
|
||||
|
||||
import { existsSync } from 'fs';
|
||||
import { basename, isAbsolute, join, resolve } from 'path';
|
||||
import pino from 'pino';
|
||||
import pinoPretty from 'pino-pretty';
|
||||
import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import {
|
||||
bannerize,
|
||||
checkTaskInfraIntegrity,
|
||||
formatIntegrityReport,
|
||||
writeManagedStamp,
|
||||
} from './lib/task-infra-integrity.js';
|
||||
import {
|
||||
checkToolkitScriptIntegrity,
|
||||
scriptIntegrityNotice,
|
||||
} from './lib/toolkit-script-integrity.js';
|
||||
|
||||
const argv = yargs(hideBin(process.argv))
|
||||
.usage('Usage: $0 <task> [options]')
|
||||
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
|
||||
.option('json', {
|
||||
type: 'boolean',
|
||||
describe: 'Output structured JSON logs',
|
||||
default: false,
|
||||
})
|
||||
.option('stamp', {
|
||||
type: 'boolean',
|
||||
default: false,
|
||||
describe:
|
||||
'Record the managed files as created, so later edits are detectable. No-op if already stamped.',
|
||||
})
|
||||
.demandCommand(1, 'Provide a task slug or directory')
|
||||
.help()
|
||||
.parseSync();
|
||||
|
||||
const log = pino(
|
||||
{ name: 'check-task-infra', level: 'info' },
|
||||
argv.json
|
||||
? process.stdout
|
||||
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
|
||||
);
|
||||
|
||||
const arg = String(argv._[0]);
|
||||
const toolkitRoot = process.cwd();
|
||||
// Accept both a bare slug and a path, since harbor-run is invoked with a path
|
||||
// (`scripts/harbor-run harbor-tasks/<slug>`) and submit-task with a slug.
|
||||
const taskDir = isAbsolute(arg)
|
||||
? arg
|
||||
: existsSync(resolve(toolkitRoot, arg))
|
||||
? resolve(toolkitRoot, arg)
|
||||
: join(toolkitRoot, 'harbor-tasks', arg);
|
||||
|
||||
if (!existsSync(taskDir)) {
|
||||
log.fatal({ taskDir }, 'Task directory not found');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const slug = basename(taskDir);
|
||||
// --stamp records a task's baseline. Runs at task creation; never overwrites.
|
||||
if (argv.stamp) {
|
||||
if (!existsSync(join(toolkitRoot, 'task-shared'))) {
|
||||
log.debug('Not a worker toolkit (no task-shared/); nothing to stamp');
|
||||
process.exit(0);
|
||||
}
|
||||
const wrote = writeManagedStamp(taskDir, toolkitRoot);
|
||||
log.debug({ slug, wrote }, wrote ? 'Stamped toolkit-managed files' : 'Already stamped');
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Toolkit scripts, not this task's files: checked here because this is already the
|
||||
// preflight both `harbor-run` and `submit-task.ts` reach. An edited script ships no
|
||||
// trace of itself, only its output — see toolkit-script-integrity.ts.
|
||||
const scripts = checkToolkitScriptIntegrity(toolkitRoot);
|
||||
if (scripts.checked) {
|
||||
const notice = scriptIntegrityNotice(scripts);
|
||||
if (notice) {
|
||||
log.warn(
|
||||
{
|
||||
edited: scripts.modified.map((f) => f.path),
|
||||
missing: scripts.missing.map((f) => f.path),
|
||||
},
|
||||
'Toolkit scripts need a look'
|
||||
);
|
||||
process.stderr.write(`\n${notice}\n\n`);
|
||||
} else {
|
||||
log.info({ files: scripts.files.length }, 'Toolkit scripts are unmodified');
|
||||
}
|
||||
}
|
||||
|
||||
const report = checkTaskInfraIntegrity(taskDir, toolkitRoot);
|
||||
|
||||
if (!report.checked) {
|
||||
log.debug(
|
||||
'Managed-file check skipped: no task-shared/ here, or the task was authored on a different toolkit generation (its tests/ assets are its own)'
|
||||
);
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const message = formatIntegrityReport(report);
|
||||
|
||||
if (!message) {
|
||||
log.info({ files: report.files.length }, 'Toolkit-managed files are unmodified');
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Advisory, always. Exiting non-zero here is what used to let a false positive stop
|
||||
// an author's trial with no way out; the report is the whole product.
|
||||
log.warn(
|
||||
{
|
||||
edited: report.modified.map((f) => f.taskPath),
|
||||
outdated: report.outdated.map((f) => f.taskPath),
|
||||
unverifiable: report.unverifiable.map((f) => f.taskPath),
|
||||
},
|
||||
'Toolkit-managed files need a look'
|
||||
);
|
||||
process.stderr.write(`\n${bannerize(message, report)}\n\n`);
|
||||
process.exit(0);
|
||||
@@ -0,0 +1,291 @@
|
||||
#!/bin/bash
|
||||
# Check that a task's live environment/workspace matches what a rebuild from
|
||||
# the pinned commit + environment/workspace.patch would produce — i.e. the
|
||||
# workspace every downstream consumer of the task actually sees. Files edited
|
||||
# (or added/deleted) directly in the built workspace are visible to your local
|
||||
# trials but do NOT survive packaging: your own tarball may carry them, but
|
||||
# the finalized task keeps only the rebuild inputs (the workspace/ dir itself
|
||||
# is gitignored), and everywhere downstream the workspace is rebuilt from the
|
||||
# gitref in task.toml plus workspace.patch (see build-workspace.sh) — anything
|
||||
# not captured there is silently dropped.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/check-workspace-sync.sh <task-dir> # check (advisory)
|
||||
# bash scripts/check-workspace-sync.sh --update-patch <task-dir> # fold live edits into workspace.patch
|
||||
#
|
||||
# Check mode is run automatically at the start of every `scripts/harbor-run`.
|
||||
# It warns loudly when the workspace has uncaptured changes, and always exits
|
||||
# 0 — it never blocks a run. It also exits 0 (silently) when it can't resolve
|
||||
# the source repo or the pinned commit, since it can't tell anything useful
|
||||
# then.
|
||||
#
|
||||
# --update-patch regenerates environment/workspace.patch as the full diff from
|
||||
# the pinned commit to the live workspace (the previous patch's changes are
|
||||
# preserved — they're part of that diff). After updating the patch, re-run
|
||||
# your trials: reference runs should be captured against the workspace every
|
||||
# downstream rebuild produces.
|
||||
#
|
||||
# Mechanics: the pinned commit's tree is read into a THROWAWAY git index (with
|
||||
# a throwaway object directory layered over the repo's, so the source repo is
|
||||
# never written to), workspace.patch is applied to that index, and the live
|
||||
# workspace directory is compared against it. Files matched by the repo's
|
||||
# .gitignore are not considered — they can't be captured in workspace.patch
|
||||
# either, so they never ship either way. File-mode-only changes are ignored
|
||||
# (core.fileMode=false), matching how patches are generated here.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
MODE="check"
|
||||
if [ "${1:-}" = "--update-patch" ]; then
|
||||
MODE="update"
|
||||
shift
|
||||
fi
|
||||
|
||||
if [ -z "${1:-}" ]; then
|
||||
echo "Usage: $0 [--update-patch] <task-dir>" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Normalize the task dir (tolerates relative paths and trailing slashes).
|
||||
TASK_DIR="$(cd "$1" 2>/dev/null && pwd)" || {
|
||||
echo "Error: task directory not found: $1" >&2
|
||||
exit 1
|
||||
}
|
||||
SLUG="$(basename "$TASK_DIR")"
|
||||
# Tasks live at <root>/harbor-tasks/<slug> in every layout this script ships to.
|
||||
ROOT="$(cd "$TASK_DIR/../.." && pwd)"
|
||||
|
||||
WORKSPACE="$TASK_DIR/environment/workspace"
|
||||
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
|
||||
TASK_TOML="$TASK_DIR/task.toml"
|
||||
|
||||
# How to spell this script in the recommendations we print. In a packed
|
||||
# toolkit it lives at <root>/scripts/ (the worker's usual cwd is <root>), so
|
||||
# the short form works; anywhere else (e.g. invoked from the internal repo
|
||||
# layout via harbor-run) fall back to the invoked path.
|
||||
SELF_DISPLAY="bash scripts/check-workspace-sync.sh"
|
||||
if [ ! -f "$ROOT/scripts/check-workspace-sync.sh" ]; then
|
||||
SELF_DISPLAY="bash $0"
|
||||
fi
|
||||
|
||||
# In check mode every "can't verify" path exits 0 quietly: this is an advisory
|
||||
# preflight and a task we can't reason about must never break a run. In
|
||||
# --update-patch mode the same conditions are hard errors — the user asked for
|
||||
# a patch and we can't produce one.
|
||||
skip() {
|
||||
if [ "$MODE" = "update" ]; then
|
||||
echo "Error: $1" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
}
|
||||
|
||||
[ -d "$WORKSPACE" ] || skip "workspace not built at $WORKSPACE (run build-workspace.sh first)"
|
||||
[ -f "$TASK_TOML" ] || skip "no task.toml at $TASK_TOML"
|
||||
|
||||
# Pinned commit: the `commit = "..."` line in task.toml. Anchored to the line
|
||||
# start so prose mentions (e.g. a `source = "... commit abc"` note) don't
|
||||
# match. No commit line is legitimate for some internally-built tasks — then
|
||||
# there's nothing to compare against.
|
||||
COMMIT="$(grep -E '^[[:space:]]*commit[[:space:]]*=' "$TASK_TOML" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)"
|
||||
[ -n "$COMMIT" ] || skip "no commit pinned in task.toml"
|
||||
|
||||
# The member this task targets, per task.toml ([metadata].repo) — used to
|
||||
# resolve the source repo in polyglot layouts. Same extraction as
|
||||
# build-workspace.sh.
|
||||
MEMBER="$(grep -E '^repo[[:space:]]*=' "$TASK_TOML" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)"
|
||||
|
||||
# Source repo resolution, in order:
|
||||
# <root>/repo — single-repo toolkit
|
||||
# <root>/repos/<member> — polyglot toolkit
|
||||
# <root>/repos/<member>/repo — internal submodule layout
|
||||
REPO_DIR=""
|
||||
for cand in "$ROOT/repo" ${MEMBER:+"$ROOT/repos/$MEMBER" "$ROOT/repos/$MEMBER/repo"}; do
|
||||
if [ -e "$cand/.git" ]; then
|
||||
REPO_DIR="$cand"
|
||||
break
|
||||
fi
|
||||
done
|
||||
[ -n "$REPO_DIR" ] || skip "source repo not found under $ROOT"
|
||||
|
||||
# Resolve the repo's real git dir (handles submodules, whose .git is a file).
|
||||
GITDIR="$(git -C "$REPO_DIR" rev-parse --absolute-git-dir 2>/dev/null)" || skip "not a git repo: $REPO_DIR"
|
||||
RESOLVED_SHA="$(git --git-dir="$GITDIR" rev-parse --quiet --verify "$COMMIT^{commit}" 2>/dev/null)" || \
|
||||
skip "pinned commit $COMMIT not found in $REPO_DIR"
|
||||
|
||||
# --- Throwaway git state ------------------------------------------------------
|
||||
# A temp index + temp object dir (with the real object dir as a read-only
|
||||
# alternate) lets us build "commit + patch" as an index and diff the live
|
||||
# workspace against it without ever writing to the source repo or creating a
|
||||
# .git inside the workspace.
|
||||
TMP="$(mktemp -d)"
|
||||
trap 'rm -rf "$TMP"' EXIT
|
||||
export GIT_INDEX_FILE="$TMP/index"
|
||||
export GIT_OBJECT_DIRECTORY="$TMP/objects"
|
||||
export GIT_ALTERNATE_OBJECT_DIRECTORIES="$GITDIR/objects"
|
||||
mkdir -p "$GIT_OBJECT_DIRECTORY"
|
||||
|
||||
# Suppress mode-bit and line-ending munging so the comparison is about content,
|
||||
# and keep non-ASCII paths readable instead of C-quoted ("\360\237...").
|
||||
GIT_FLAGS=(-c core.fileMode=false -c core.autocrlf=false -c core.quotePath=false)
|
||||
|
||||
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
|
||||
|
||||
# `.zeta-siblings/` is staged INTO the workspace by build-workspace.sh on some
|
||||
# toolkits (bundled sibling deps) — a build artifact, never part of the patch.
|
||||
# `.raccoon-setup-done` is run-app's per-repo first-use setup marker (polyglot
|
||||
# toolkits) — authoring-machine state, never task content. run-app git-ignores
|
||||
# it via the repo's .git/info/exclude, but this script diffs through a
|
||||
# throwaway --git-dir that never reads that file, so exclude it here too.
|
||||
# The leading `.` positive pathspec is load-bearing: several git commands
|
||||
# reject a pathspec made of nothing but exclusions.
|
||||
EXCLUDES=("." ":(exclude).zeta-siblings" ":(exclude).raccoon-setup-done")
|
||||
|
||||
cd "$WORKSPACE"
|
||||
export GIT_WORK_TREE="$WORKSPACE"
|
||||
|
||||
if [ "$MODE" = "update" ]; then
|
||||
# Stage the live workspace on top of the pinned tree, then emit the full
|
||||
# tree -> index diff as the new workspace.patch. --binary --full-index so
|
||||
# binary additions survive a later `git apply`.
|
||||
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" add -A -- "${EXCLUDES[@]}"
|
||||
NEW_PATCH="$TMP/workspace.patch"
|
||||
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --binary --full-index "$RESOLVED_SHA" -- "${EXCLUDES[@]}" > "$NEW_PATCH"
|
||||
|
||||
if [ ! -s "$NEW_PATCH" ]; then
|
||||
if [ -f "$PATCH_FILE" ]; then
|
||||
rm -f "$PATCH_FILE"
|
||||
echo "Workspace matches commit $COMMIT exactly — removed the now-empty environment/workspace.patch."
|
||||
else
|
||||
echo "Workspace matches commit $COMMIT exactly — no workspace.patch needed."
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Verify the regenerated patch applies to the pristine tree before
|
||||
# installing it, so we never leave behind a patch build-workspace.sh
|
||||
# would choke on.
|
||||
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
|
||||
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached --check "$NEW_PATCH" || {
|
||||
echo "Error: regenerated patch does not apply cleanly to $COMMIT — workspace.patch left unchanged." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
cp "$NEW_PATCH" "$PATCH_FILE"
|
||||
FILE_COUNT="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --name-only "$RESOLVED_SHA" -- "${EXCLUDES[@]}" | wc -l | tr -d ' ')"
|
||||
echo "Wrote environment/workspace.patch: $FILE_COUNT file(s) differ from commit $COMMIT."
|
||||
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
|
||||
echo ""
|
||||
echo "NOTE: this task was built from a session snapshot, so workspace.patch is"
|
||||
echo "meant to mirror the workspace state the captured session describes. Make"
|
||||
echo "sure these folded-in changes don't contradict the session transcript —"
|
||||
echo "if they belong to the session's story, re-capturing the snapshot"
|
||||
echo "(/create-snapshot:snapshot, then scripts/snapshot-to-task.ts) is the"
|
||||
echo "cleaner fix."
|
||||
fi
|
||||
echo ""
|
||||
echo "Re-run your trials so your reference runs match what now ships:"
|
||||
echo " scripts/harbor-run harbor-tasks/$SLUG -k 4"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# --- Check mode ---------------------------------------------------------------
|
||||
|
||||
# Apply workspace.patch to the throwaway index — the index then holds exactly
|
||||
# the tree build-workspace.sh would produce. A patch that no longer applies is
|
||||
# its own (serious) problem: the shipped inputs can't even rebuild.
|
||||
if [ -s "$PATCH_FILE" ]; then
|
||||
if ! git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached "$PATCH_FILE" 2>/dev/null; then
|
||||
echo "" >&2
|
||||
echo "==============================================================================" >&2
|
||||
echo "!! WARNING: environment/workspace.patch does not apply to commit $COMMIT." >&2
|
||||
echo "!! A rebuild of this task from its shipped inputs (build-workspace.sh)" >&2
|
||||
echo "!! would FAIL, and your live workspace can't be checked against them." >&2
|
||||
echo "!! Did the gitref or the patch change after the workspace was built?" >&2
|
||||
echo "==============================================================================" >&2
|
||||
echo "" >&2
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# Tracked files that differ between the index (commit + patch) and the live
|
||||
# workspace, plus files that exist only in the live workspace. Both respect
|
||||
# the repo's .gitignore.
|
||||
DIFF_RAW="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --name-status -- "${EXCLUDES[@]}")"
|
||||
UNTRACKED="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" ls-files --others --exclude-standard -- "${EXCLUDES[@]}")"
|
||||
|
||||
# Deletions of symlinks are skipped: the internal workspace build prunes
|
||||
# dangling symlinks after applying the patch, so their absence is expected,
|
||||
# not a worker edit.
|
||||
CHANGES=""
|
||||
while IFS=$'\t' read -r st path; do
|
||||
[ -n "$st" ] || continue
|
||||
if [ "$st" = "D" ]; then
|
||||
entry_mode="$(git --git-dir="$GITDIR" ls-files -s -- "$path" | awk '{print $1}')"
|
||||
[ "$entry_mode" = "120000" ] && continue
|
||||
fi
|
||||
CHANGES="${CHANGES} ${st} ${path}
|
||||
"
|
||||
done <<< "$DIFF_RAW"
|
||||
while IFS= read -r path; do
|
||||
[ -n "$path" ] || continue
|
||||
CHANGES="${CHANGES} ?? ${path}
|
||||
"
|
||||
done <<< "$UNTRACKED"
|
||||
|
||||
[ -n "$CHANGES" ] || exit 0
|
||||
|
||||
TOTAL="$(printf '%s' "$CHANGES" | wc -l | tr -d ' ')"
|
||||
LISTED="$(printf '%s' "$CHANGES" | head -25)"
|
||||
|
||||
{
|
||||
echo ""
|
||||
echo "=============================================================================="
|
||||
echo "!! WARNING: environment/workspace has changes that will NOT survive"
|
||||
echo "!! packaging."
|
||||
echo "=============================================================================="
|
||||
echo ""
|
||||
echo "The workspace/ directory itself is never kept: everywhere downstream the"
|
||||
echo "task is rebuilt from the commit pinned in task.toml ($COMMIT) plus"
|
||||
echo "environment/workspace.patch — exactly what scripts/build-workspace.sh"
|
||||
echo "produces. These $TOTAL file(s) differ from that rebuild, so your local trials"
|
||||
echo "see them, but they will not survive packaging:"
|
||||
echo ""
|
||||
echo "$LISTED"
|
||||
if [ "$TOTAL" -gt 25 ]; then
|
||||
echo " ... and $((TOTAL - 25)) more"
|
||||
fi
|
||||
echo ""
|
||||
echo " (M = modified, D = deleted, ?? = only in the live workspace. Files matched"
|
||||
echo " by the repo's .gitignore are not checked — they never ship either way.)"
|
||||
echo ""
|
||||
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
|
||||
echo "This task was built from a session snapshot, and workspace.patch mirrors"
|
||||
echo "the workspace state captured with that session. If these changes belong in"
|
||||
echo "the task, the cleanest fix is to make them in the Explore session and"
|
||||
echo "re-capture (/create-snapshot:snapshot, then scripts/snapshot-to-task.ts),"
|
||||
echo "so the session transcript and the workspace stay consistent."
|
||||
echo ""
|
||||
echo "To fold them into workspace.patch anyway — only if they don't contradict"
|
||||
echo "what the captured session says about the workspace:"
|
||||
echo ""
|
||||
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
|
||||
else
|
||||
echo "To fold these changes into workspace.patch so they ship with the task:"
|
||||
echo ""
|
||||
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
|
||||
if [ -f "$ROOT/scripts/build-workspace.sh" ]; then
|
||||
echo ""
|
||||
echo "To discard them instead (rebuild the workspace from commit + patch):"
|
||||
echo ""
|
||||
echo " bash scripts/build-workspace.sh $SLUG"
|
||||
fi
|
||||
fi
|
||||
echo ""
|
||||
echo "Either way, re-run your trials afterwards so your reference runs match the"
|
||||
echo "workspace every downstream rebuild produces."
|
||||
echo "=============================================================================="
|
||||
echo ""
|
||||
} >&2
|
||||
|
||||
exit 0
|
||||
File diff suppressed because one or more lines are too long
545
worker-toolkit-potion-polyglot-orig/scripts/codex_agent.py
Normal file
545
worker-toolkit-potion-polyglot-orig/scripts/codex_agent.py
Normal file
@@ -0,0 +1,545 @@
|
||||
"""Custom Codex agents for our devcontainer-based task images.
|
||||
|
||||
Harbor's stock Codex agent (`harbor.agents.installed.codex.Codex`) installs Node
|
||||
via nvm into `$HOME/.nvm` during `install()`, and `setup()` ALWAYS calls
|
||||
`install()` (the version probe only runs afterward). That install fails on our
|
||||
task images: they are `FROM mcr.microsoft.com/devcontainers/typescript-node:20`,
|
||||
which provides Node through the devcontainer nvm at `/usr/local/share/nvm`, and
|
||||
the tasks run as root, where that nvm isn't auto-loaded — so harbor's
|
||||
`$HOME/.nvm/nvm.sh` doesn't exist and the agent dies with "NVM failed to load".
|
||||
|
||||
`SystemNodeCodex` overrides `install()` to load the image's existing Node and
|
||||
install only the codex CLI (no second Node via nvm). Everything else — the
|
||||
trajectory parsing, the codex exec, reasoning_effort kwargs — is inherited
|
||||
unchanged from the stock agent.
|
||||
|
||||
Use via: `--agent-import-path codex_agent:SystemNodeCodex` (PYTHONPATH=scripts).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shlex
|
||||
import sys
|
||||
import tempfile
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
import atif_session
|
||||
import browser_note
|
||||
try:
|
||||
from dnsjail import apply_dns_jail
|
||||
except ImportError: # no helper shipped -> no jail, rather than no trials
|
||||
|
||||
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
|
||||
return None
|
||||
|
||||
|
||||
from harbor.agents.installed.codex import Codex
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
|
||||
|
||||
import codex_auth # noqa: E402
|
||||
from harness_registry import load_harness_registry # noqa: E402
|
||||
|
||||
# Installs the codex CLI into /usr/local/bin so harbor's plain-sh execs find it.
|
||||
#
|
||||
# Primary path is the official standalone installer, which fetches a prebuilt
|
||||
# native binary and needs only curl + tar — no Node in the image. That matters
|
||||
# because most task images (Ruby/Python) ship no Node at all, and the npm route
|
||||
# below can only run on the Node-bearing minority.
|
||||
#
|
||||
# The npm route is kept as a fallback for images where the installer can't run
|
||||
# (e.g. a native binary the image's glibc rejects) but a usable npm exists.
|
||||
_INSTALL_CMD = (
|
||||
"set -x; "
|
||||
"if command -v apt-get >/dev/null 2>&1; then "
|
||||
" apt-get update -qq >/dev/null 2>&1 && apt-get install -y -qq curl ripgrep >/dev/null 2>&1 || true; "
|
||||
"fi; "
|
||||
'if ! command -v codex >/dev/null 2>&1; then '
|
||||
' CODEX_INSTALL_DIR=/usr/local/bin CODEX_NON_INTERACTIVE=true '
|
||||
' sh -c "curl -fsSL https://chatgpt.com/codex/install.sh | sh" >&2 || true; '
|
||||
"fi; "
|
||||
'if ! command -v codex >/dev/null 2>&1 && [ -x "$HOME/.local/bin/codex" ]; then '
|
||||
' ln -sf "$HOME/.local/bin/codex" /usr/local/bin/codex; '
|
||||
"fi; "
|
||||
# npm fallback: load the devcontainer nvm, else find npm anywhere plausible.
|
||||
'if ! command -v codex >/dev/null 2>&1; then '
|
||||
' export NVM_DIR="${NVM_DIR:-/usr/local/share/nvm}"; '
|
||||
' [ -s "$NVM_DIR/nvm.sh" ] && . "$NVM_DIR/nvm.sh" >/dev/null 2>&1 || true; '
|
||||
' if ! command -v npm >/dev/null 2>&1; then '
|
||||
' npm_path="$(find /usr/local/share/nvm /usr/local /usr/lib /opt -name npm -type f 2>/dev/null | head -1)"; '
|
||||
' [ -n "$npm_path" ] && export PATH="$PATH:$(dirname "$npm_path")"; '
|
||||
" fi; "
|
||||
' command -v npm >/dev/null 2>&1 && npm install -g @openai/codex@latest; '
|
||||
"fi; "
|
||||
'for bin in node codex; do '
|
||||
' p="$(command -v "$bin" 2>/dev/null || true)"; '
|
||||
' [ -n "$p" ] && [ "$p" != "/usr/local/bin/$bin" ] && ln -sf "$p" "/usr/local/bin/$bin" || true; '
|
||||
"done; "
|
||||
'command -v codex >/dev/null 2>&1 '
|
||||
' || { echo "FATAL: codex CLI unavailable (standalone installer and npm both failed)" >&2; exit 1; }; '
|
||||
"codex --version"
|
||||
)
|
||||
|
||||
|
||||
class SystemNodeCodex(Codex):
|
||||
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
|
||||
_has_browser = False
|
||||
|
||||
# Non-snapshot codex tasks run this class directly; the native-snapshot resume path
|
||||
# overrides run() and applies the jail itself.
|
||||
async def run(self, instruction, environment, context): # type: ignore[override]
|
||||
self._refuse_shell_hostile_key()
|
||||
await apply_dns_jail(self, environment)
|
||||
await super().run(instruction, environment, context)
|
||||
|
||||
def _auth_json_setup(self, remote_auth_path: str) -> tuple[dict[str, str], str]:
|
||||
return codex_auth.auth_json_setup(
|
||||
self._get_env("OPENAI_API_KEY") or "", remote_auth_path
|
||||
)
|
||||
|
||||
def _refuse_shell_hostile_key(self) -> None:
|
||||
"""Harbor's own Codex.run interpolates the key into a heredoc, so a key it cannot
|
||||
escape would 401 with no stated cause. Refuse up front instead."""
|
||||
if self._resolve_auth_json_path():
|
||||
return
|
||||
bad = codex_auth.unescapable_chars(self._get_env("OPENAI_API_KEY") or "")
|
||||
if bad:
|
||||
raise ValueError(
|
||||
"OPENAI_API_KEY contains "
|
||||
+ ", ".join(repr(c) for c in bad)
|
||||
+ ", which harbor's stock auth.json writer cannot escape. Point "
|
||||
"CODEX_AUTH_JSON_PATH at a pre-written auth.json instead."
|
||||
)
|
||||
|
||||
async def install(self, environment) -> None: # type: ignore[override]
|
||||
await self.exec_as_root(environment, command=_INSTALL_CMD)
|
||||
self._has_browser = await browser_note.probe_browser(environment)
|
||||
|
||||
def build_cli_flags(self) -> str: # type: ignore[override]
|
||||
"""Harbor's flags plus the registry's `agent_config`, so a trial's toolset
|
||||
matches the explore launcher's — which passes the same rendering as
|
||||
$RACCOON_AGENT_FLAGS. Both run paths go through here."""
|
||||
flags = super().build_cli_flags()
|
||||
reductions = load_harness_registry().require("codex").agent_config_flags()
|
||||
if reductions:
|
||||
flags = f"{flags} {reductions}".strip()
|
||||
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
|
||||
|
||||
def _browser_flag(self) -> str:
|
||||
"""Disclose the browser to codex the way codex takes extra instructions.
|
||||
|
||||
`developer_instructions` PREPENDS a developer message and leaves codex's own base
|
||||
instructions in place — verified with `codex debug prompt-input`. That makes it the
|
||||
equivalent of claude's --append-system-prompt. `model_instructions_file`, the other
|
||||
instruction-shaped key, REPLACES the base instructions; do not use it here.
|
||||
"""
|
||||
if not self._has_browser:
|
||||
return ""
|
||||
note = browser_note.browser_note()
|
||||
if not note:
|
||||
return ""
|
||||
return f"-c developer_instructions={shlex.quote(note)}"
|
||||
|
||||
|
||||
# Where harbor's run-prep stages the prior Claude Code session for snapshot tasks
|
||||
# (the same file our snapshot_agent reads). Agent-agnostic, so codex sees it too.
|
||||
_STAGED_SESSION = "/tmp/snapshot-session/session.jsonl"
|
||||
|
||||
|
||||
def render_claude_session(jsonl_text: str, max_block: int = 4000) -> str:
|
||||
"""Render a Claude Code session JSONL transcript into readable plain text so a
|
||||
non-Claude agent (codex) can be handed the prior conversation as context.
|
||||
|
||||
Each line is a Claude record: {"type": "user"|"assistant", "message": {"role",
|
||||
"content"}}. `content` is either a string or a list of blocks
|
||||
(text / tool_use / tool_result / thinking). We flatten to labeled turns and
|
||||
truncate oversized tool payloads so the context stays bounded."""
|
||||
out: list[str] = []
|
||||
|
||||
def clip(s: str) -> str:
|
||||
s = s.rstrip()
|
||||
return s if len(s) <= max_block else s[:max_block] + "\n…[truncated]"
|
||||
|
||||
for line in jsonl_text.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
rtype = rec.get("type")
|
||||
msg = rec.get("message") or {}
|
||||
role = msg.get("role") or rtype
|
||||
content = msg.get("content")
|
||||
if content is None:
|
||||
# non-message records (summaries, etc.) — skip unless they carry text
|
||||
txt = rec.get("summary") or rec.get("content")
|
||||
if isinstance(txt, str) and txt.strip():
|
||||
out.append(f"[{rtype}] {clip(txt)}")
|
||||
continue
|
||||
if isinstance(content, str):
|
||||
out.append(f"{role.upper()}: {clip(content)}")
|
||||
continue
|
||||
# content is a list of blocks
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
out.append(f"{role.upper()}: {clip(str(block))}")
|
||||
continue
|
||||
btype = block.get("type")
|
||||
if btype == "text":
|
||||
out.append(f"{role.upper()}: {clip(block.get('text', ''))}")
|
||||
elif btype == "thinking":
|
||||
out.append(f"{role.upper()} (thinking): {clip(block.get('thinking', ''))}")
|
||||
elif btype == "tool_use":
|
||||
name = block.get("name", "?")
|
||||
inp = json.dumps(block.get("input", {}), ensure_ascii=False)
|
||||
out.append(f"{role.upper()} [tool_use {name}]: {clip(inp)}")
|
||||
elif btype == "tool_result":
|
||||
res = block.get("content")
|
||||
if isinstance(res, list):
|
||||
res = "".join(
|
||||
b.get("text", "") for b in res if isinstance(b, dict)
|
||||
)
|
||||
out.append(f"[tool_result]: {clip(str(res))}")
|
||||
return "\n".join(out)
|
||||
|
||||
|
||||
_INLINE_PREAMBLE = (
|
||||
"You are continuing an in-progress pair-programming session. Below is the FULL "
|
||||
"prior conversation between the user and the previous assistant (you), including "
|
||||
"the tool calls that assistant made and their results. Treat it as your own prior "
|
||||
"context — the workspace already reflects any edits made in it. Then respond to the "
|
||||
"user's newest message at the end.\n\n"
|
||||
"================ PRIOR CONVERSATION ================\n"
|
||||
)
|
||||
|
||||
|
||||
class InlineSnapshotCodex(SystemNodeCodex):
|
||||
"""Bridge A: run codex on snapshot tasks by INLINING the prior Claude session as
|
||||
plain-text context ahead of the user's next-turn instruction. Works for any
|
||||
provider — codex just sees a long prompt: [rendered prior conversation] + [the
|
||||
user's newest message]. For non-snapshot tasks (no staged session) it behaves
|
||||
exactly like the stock codex agent."""
|
||||
|
||||
async def run(self, instruction, environment, context): # type: ignore[override]
|
||||
session_text = ""
|
||||
try:
|
||||
result = await environment.exec(
|
||||
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
|
||||
)
|
||||
session_text = (getattr(result, "stdout", "") or "").strip()
|
||||
except Exception as exc: # best-effort; fall back to bare instruction
|
||||
self.logger.warning("InlineSnapshotCodex: could not read session: %s", exc)
|
||||
|
||||
if session_text:
|
||||
rendered = render_claude_session(session_text)
|
||||
if rendered.strip():
|
||||
instruction = (
|
||||
_INLINE_PREAMBLE
|
||||
+ rendered
|
||||
+ "\n\n================ USER'S NEWEST MESSAGE ================\n"
|
||||
+ instruction
|
||||
)
|
||||
self.logger.info(
|
||||
"InlineSnapshotCodex: injected %d chars of rendered prior session",
|
||||
len(rendered),
|
||||
)
|
||||
else:
|
||||
self.logger.warning("InlineSnapshotCodex: session rendered empty")
|
||||
else:
|
||||
self.logger.info(
|
||||
"InlineSnapshotCodex: no staged session (non-snapshot task or empty); "
|
||||
"running bare instruction"
|
||||
)
|
||||
await super().run(instruction, environment, context)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bridge B: native codex resume.
|
||||
#
|
||||
# Instead of inlining the whole prior Claude session into one giant prompt
|
||||
# (Bridge A, which makes codex stall on a ~50k-token blob), we translate the
|
||||
# staged session into codex's OWN rollout JSONL format, drop it into
|
||||
# $CODEX_HOME/sessions/<date>/rollout-<ts>-<uuid>.jsonl, and invoke
|
||||
# `codex exec resume <uuid> -- <instruction>`. codex then treats the prior turns
|
||||
# as its own conversation history — prompt-cached and incremental — and only has
|
||||
# to reason about the user's newest message.
|
||||
#
|
||||
# We resume by EXPLICIT session id (not --last): --last is cwd-filtered (help:
|
||||
# "--all ... disables cwd filtering"), and we can't guarantee the rollout's
|
||||
# recorded cwd matches the sandbox cwd at runtime; an explicit UUID is a direct
|
||||
# lookup that sidesteps that entirely.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# A real recorded codex session_meta line (with codex's base_instructions) is the
|
||||
# most reliable seed for `resume`. The fixture lives in-repo (mounted into the
|
||||
# devcontainer where this agent code runs); a captured host copy is a secondary
|
||||
# source, and a synthesized minimal record is the final fallback.
|
||||
_ROLLOUT_TEMPLATE_CANDIDATES = (
|
||||
os.path.join(os.path.dirname(os.path.abspath(__file__)), "codex-rollout-template.jsonl"),
|
||||
"/Users/nickheiner/.claude/jobs/e8fade29/tmp/codex-rollout-template.jsonl",
|
||||
)
|
||||
|
||||
|
||||
def _now_iso() -> str:
|
||||
from datetime import datetime, timezone
|
||||
|
||||
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z")
|
||||
|
||||
|
||||
def _session_meta(new_id: str, iso_ts: str) -> dict:
|
||||
"""Return a codex `session_meta` rollout record, reusing the captured real
|
||||
template (best fidelity for resume) when readable, else a minimal synthesized
|
||||
one. The id/timestamp are always overwritten with our fresh values."""
|
||||
for template_path in _ROLLOUT_TEMPLATE_CANDIDATES:
|
||||
try:
|
||||
with open(template_path, encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
rec = json.loads(line)
|
||||
if rec.get("type") == "session_meta":
|
||||
rec["timestamp"] = iso_ts
|
||||
rec.setdefault("payload", {})
|
||||
rec["payload"]["id"] = new_id
|
||||
rec["payload"]["timestamp"] = iso_ts
|
||||
return rec
|
||||
except (OSError, ValueError):
|
||||
continue
|
||||
return {
|
||||
"timestamp": iso_ts,
|
||||
"type": "session_meta",
|
||||
"payload": {
|
||||
"id": new_id,
|
||||
"timestamp": iso_ts,
|
||||
"cwd": "/workspace",
|
||||
"originator": "codex_exec",
|
||||
"cli_version": "0.135.0",
|
||||
"source": "exec",
|
||||
"thread_source": "user",
|
||||
"model_provider": "openai",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
|
||||
def is_codex_rollout(session_jsonl_text: str) -> bool:
|
||||
"""True when the staged session is already a codex rollout rather than a Claude Code
|
||||
transcript. Delegates to atif_session, which owns format detection — a second copy of
|
||||
the record-type set here is how the two would eventually disagree."""
|
||||
return atif_session.detect_format(session_jsonl_text) == "codex"
|
||||
|
||||
def reid_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
|
||||
"""Re-key an already-native codex rollout onto `new_id` so `codex exec resume
|
||||
<new_id>` finds it. Conversation records pass through byte-identical — a
|
||||
codex-authored snapshot resumed by codex needs no translation, which is the
|
||||
whole fidelity argument for native seeding."""
|
||||
lines = [json.dumps(_session_meta(new_id, iso_ts))]
|
||||
for raw in session_jsonl_text.splitlines():
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
rec = json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
if not isinstance(rec, dict) or rec.get("type") == "session_meta":
|
||||
continue
|
||||
lines.append(raw)
|
||||
return lines
|
||||
|
||||
|
||||
def stage_to_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
|
||||
"""Build a resumable codex rollout from whichever format the snapshot staged."""
|
||||
if is_codex_rollout(session_jsonl_text):
|
||||
return reid_codex_rollout(session_jsonl_text, new_id, iso_ts)
|
||||
return claude_to_codex_rollout(session_jsonl_text, new_id, iso_ts)
|
||||
|
||||
|
||||
def claude_to_codex_rollout(
|
||||
session_jsonl_text: str, new_id: str, iso_ts: str, max_total: int | None = None
|
||||
) -> list[str]:
|
||||
"""Translate a staged Claude Code session into codex rollout JSONL lines so
|
||||
`codex exec resume` can continue it natively.
|
||||
|
||||
Parsing and rendering live in atif_session, which routes every harness pair
|
||||
through ATIF; this stays as the codex-side entry point. `max_total` is an optional
|
||||
char cap used by tests; the default is uncapped — our seeded sessions (~20-95k
|
||||
tokens) fit every supported model's context."""
|
||||
return atif_session.atif_to_codex_rollout(
|
||||
atif_session.claude_session_to_atif(session_jsonl_text),
|
||||
iso_ts,
|
||||
session_meta=_session_meta(new_id, iso_ts),
|
||||
max_total=max_total,
|
||||
)
|
||||
|
||||
|
||||
class NativeSnapshotCodex(SystemNodeCodex):
|
||||
"""Bridge B: continue the staged Claude session via NATIVE codex resume.
|
||||
|
||||
For snapshot tasks we translate `/tmp/snapshot-session/session.jsonl` into a
|
||||
codex rollout, write it under `$CODEX_HOME/sessions/`, and run
|
||||
`codex exec resume <uuid> -- <instruction>`. For non-snapshot tasks (no staged
|
||||
session) we defer to the stock fresh `codex exec` via the base agent."""
|
||||
|
||||
async def _read_staged_session(self, environment) -> str:
|
||||
try:
|
||||
result = await environment.exec(
|
||||
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
|
||||
)
|
||||
return (getattr(result, "stdout", "") or "").strip()
|
||||
except Exception as exc: # best-effort
|
||||
self.logger.warning("NativeSnapshotCodex: could not read session: %s", exc)
|
||||
return ""
|
||||
|
||||
async def run(self, instruction, environment, context): # type: ignore[override]
|
||||
# NOTE: codex unconditionally declares its `tool_search` (MCP apps tool-
|
||||
# discovery) tool, which the OpenAI API REJECTS for nano models with HTTP
|
||||
# 400 "Tool 'tool_search' is not supported". None of codex's knobs
|
||||
# (--disable tool_search / features.tool_search / enable_mcp_apps=false /
|
||||
# disabled_tools) suppress it as of codex 0.135, so nano models are NOT
|
||||
# runnable under this harness. Use a mini (e.g. gpt-5.4-mini) for the small
|
||||
# end instead. Non-nano models are unaffected.
|
||||
await apply_dns_jail(self, environment)
|
||||
session_text = await self._read_staged_session(environment)
|
||||
if not session_text:
|
||||
self.logger.info(
|
||||
"NativeSnapshotCodex: no staged session (non-snapshot task or empty); "
|
||||
"running stock fresh codex exec"
|
||||
)
|
||||
await SystemNodeCodex.run(self, instruction, environment, context)
|
||||
return
|
||||
|
||||
if not self.model_name:
|
||||
raise ValueError("Model name is required")
|
||||
model = self.model_name.split("/")[-1]
|
||||
|
||||
new_id = str(uuid.uuid4())
|
||||
iso_ts = _now_iso()
|
||||
rollout_lines = stage_to_codex_rollout(session_text, new_id, iso_ts)
|
||||
self.logger.info(
|
||||
"NativeSnapshotCodex: %s rollout of %d records (~%d chars) for resume %s",
|
||||
"re-keyed native" if is_codex_rollout(session_text) else "translated Claude",
|
||||
len(rollout_lines),
|
||||
sum(len(line) for line in rollout_lines),
|
||||
new_id,
|
||||
)
|
||||
|
||||
# --- auth/setup: faithful to harbor's Codex.run (OPENAI_API_KEY → auth.json) ---
|
||||
escaped_instruction = shlex.quote(instruction)
|
||||
cli_flags = self.build_cli_flags()
|
||||
cli_flags_arg = (cli_flags + " ") if cli_flags else ""
|
||||
auth_json_path = self._resolve_auth_json_path()
|
||||
remote_codex_home = self._REMOTE_CODEX_HOME.as_posix()
|
||||
remote_secrets_dir = self._REMOTE_CODEX_SECRETS_DIR.as_posix()
|
||||
remote_auth_path = (self._REMOTE_CODEX_SECRETS_DIR / "auth.json").as_posix()
|
||||
|
||||
env: dict[str, str] = {"CODEX_HOME": remote_codex_home}
|
||||
setup_env: dict[str, str] = {}
|
||||
await self.exec_as_agent(
|
||||
environment,
|
||||
command=(
|
||||
f'mkdir -p "$CODEX_HOME" {shlex.quote(remote_secrets_dir)} '
|
||||
f"{shlex.quote(EnvironmentPaths.agent_dir.as_posix())}"
|
||||
),
|
||||
env=env,
|
||||
)
|
||||
if auth_json_path:
|
||||
await environment.upload_file(auth_json_path, remote_auth_path)
|
||||
if environment.default_user is not None:
|
||||
await self.exec_as_root(
|
||||
environment,
|
||||
command=f"chown {environment.default_user} {remote_auth_path}",
|
||||
)
|
||||
setup_command = f'ln -sf {shlex.quote(remote_auth_path)} "$CODEX_HOME/auth.json"\n'
|
||||
else:
|
||||
env["OPENAI_API_KEY"] = self._get_env("OPENAI_API_KEY") or ""
|
||||
setup_env, auth_command = self._auth_json_setup(remote_auth_path)
|
||||
setup_command = (
|
||||
auth_command
|
||||
+ f"ln -sf {shlex.quote(remote_auth_path)} \"$CODEX_HOME/auth.json\"\n"
|
||||
)
|
||||
if openai_base_url := self._get_env("OPENAI_BASE_URL"):
|
||||
env["OPENAI_BASE_URL"] = openai_base_url
|
||||
setup_command += (
|
||||
'\ncat >>"$CODEX_HOME/config.toml" <<TOML\n'
|
||||
'openai_base_url = "${OPENAI_BASE_URL}"\n'
|
||||
"TOML"
|
||||
)
|
||||
skills_command = self._build_register_skills_command()
|
||||
if skills_command:
|
||||
setup_command += f"\n{skills_command}"
|
||||
mcp_command = self._build_register_mcp_servers_command()
|
||||
if mcp_command:
|
||||
setup_command += f"\n{mcp_command}"
|
||||
if setup_command.strip():
|
||||
await self.exec_as_agent(
|
||||
environment, command=setup_command, env={**env, **setup_env}
|
||||
)
|
||||
|
||||
# --- write the converted rollout into $CODEX_HOME/sessions/<date>/ ---
|
||||
date_parts = iso_ts[:10].split("-") # YYYY, MM, DD
|
||||
sessions_dir = f"{remote_codex_home}/sessions/{date_parts[0]}/{date_parts[1]}/{date_parts[2]}"
|
||||
rollout_name = f"rollout-{iso_ts.replace(':', '-')}-{new_id}.jsonl"
|
||||
remote_rollout = f"{sessions_dir}/{rollout_name}"
|
||||
await self.exec_as_agent(
|
||||
environment, command=f"mkdir -p {shlex.quote(sessions_dir)}", env=env
|
||||
)
|
||||
with tempfile.NamedTemporaryFile(
|
||||
"w", suffix=".jsonl", delete=False, encoding="utf-8"
|
||||
) as tmp:
|
||||
tmp.write("\n".join(rollout_lines) + "\n")
|
||||
host_rollout = tmp.name
|
||||
try:
|
||||
await environment.upload_file(host_rollout, remote_rollout)
|
||||
if environment.default_user is not None:
|
||||
await self.exec_as_root(
|
||||
environment,
|
||||
command=f"chown {environment.default_user} {shlex.quote(remote_rollout)}",
|
||||
)
|
||||
finally:
|
||||
try:
|
||||
os.unlink(host_rollout)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
# --- resume by explicit session id ---
|
||||
try:
|
||||
await self.exec_as_agent(
|
||||
environment,
|
||||
command=(
|
||||
"if [ -s ~/.nvm/nvm.sh ]; then . ~/.nvm/nvm.sh; fi; "
|
||||
f"codex exec resume {new_id} "
|
||||
"--dangerously-bypass-approvals-and-sandbox "
|
||||
"--skip-git-repo-check "
|
||||
f"--model {model} "
|
||||
"--json "
|
||||
"--enable unified_exec "
|
||||
f"{cli_flags_arg}"
|
||||
"-- "
|
||||
f"{escaped_instruction} "
|
||||
f"2>&1 </dev/null | tee {EnvironmentPaths.agent_dir / self._OUTPUT_FILENAME}"
|
||||
),
|
||||
env=env,
|
||||
)
|
||||
finally:
|
||||
try:
|
||||
await self.exec_as_agent(
|
||||
environment,
|
||||
command=(
|
||||
f"mkdir -p {EnvironmentPaths.agent_dir.as_posix()}\n"
|
||||
'if [ -d "$CODEX_HOME/sessions" ]; then\n'
|
||||
f" rm -rf {(EnvironmentPaths.agent_dir / 'sessions').as_posix()}\n"
|
||||
f' cp -R "$CODEX_HOME/sessions" {(EnvironmentPaths.agent_dir / "sessions").as_posix()}\n'
|
||||
"fi"
|
||||
),
|
||||
env=env,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
@@ -0,0 +1,393 @@
|
||||
/**
|
||||
* copy-reference-run.ts - Copy Harbor job trials into a task's reference-runs directory.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
|
||||
*
|
||||
* Examples:
|
||||
* # Copy a single trial
|
||||
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__3Df3Bjr
|
||||
*
|
||||
* # Copy all trials from a job
|
||||
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__*
|
||||
*
|
||||
* What gets copied:
|
||||
* - verifier/agent-output/ (answer.md, etc.)
|
||||
* - verifier/reward.txt (the run's reward) + reward-correctness.txt (reads
|
||||
* N/A by design — correctness lives inside the graded criteria)
|
||||
* - verifier/reward.json (the machine-readable reward record)
|
||||
* - verifier/signals-status.txt (whether every deterministic check actually ran,
|
||||
* i.e. whether the signals the grader was fed are complete)
|
||||
* - verifier/grade.md + every grade-<N>.md grader sample
|
||||
* - verifier/grader-result(-<N>).json, grader-stderr(-<N>).log, grader-samples.txt
|
||||
* - verifier/grader-regime.json (the grading regime this grade actually ran
|
||||
* under — unrecoverable after the fact, so it must travel with the grade)
|
||||
* - agent/claude-code.txt or agent/codex.txt (the harness's own log), agent/trajectory.json
|
||||
* - session.jsonl — the resumable session log, hoisted to the top of the run dir so
|
||||
* `view-harbor-session.ts <run-dir>/session.jsonl` can load it without further
|
||||
* indirection. Its location is per harness: Claude Code writes
|
||||
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
|
||||
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
|
||||
* - config.json, result.json, trial.log
|
||||
* - input-checksums.json — sha256 checksums of the task inputs the run was
|
||||
* generated against (prompt, session snapshot, workspace patch, gitref),
|
||||
* so submit-task.ts can warn when the run goes stale. Copied from the
|
||||
* trial dir when harbor-run stamped one at launch time (capturedBy:
|
||||
* 'run' — immune to edits made between the run and this copy);
|
||||
* otherwise captured here at copy time as a fallback (capturedBy:
|
||||
* 'copy').
|
||||
*
|
||||
* What is NOT copied:
|
||||
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
|
||||
* is hoisted out as `session.jsonl` above; everything else here is
|
||||
* untyped workspace state)
|
||||
* - artifacts/
|
||||
*/
|
||||
|
||||
import './lib/check-devcontainer';
|
||||
|
||||
import {
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
readdirSync,
|
||||
readFileSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'fs';
|
||||
import { basename, join } from 'path';
|
||||
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { copyPath, copyTree } from './lib/copy-tree';
|
||||
import {
|
||||
captureTaskInputs,
|
||||
INPUT_CHECKSUMS_FILENAME,
|
||||
readTaskInputChecksums,
|
||||
} from './lib/input-checksums';
|
||||
import { manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
|
||||
import { readSessionId } from './session-id';
|
||||
|
||||
const HARBOR_TASKS_DIR = 'harbor-tasks';
|
||||
|
||||
function findTaskDir(trialPrefix: string): string | null {
|
||||
if (!existsSync(HARBOR_TASKS_DIR)) return null;
|
||||
const entries = readdirSync(HARBOR_TASKS_DIR, { withFileTypes: true });
|
||||
for (const entry of entries) {
|
||||
if (entry.isDirectory() && entry.name.startsWith(trialPrefix)) {
|
||||
return join(HARBOR_TASKS_DIR, entry.name);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Newest `rollout-*.jsonl` anywhere under a codex `sessions/` tree, or null. */
|
||||
function newestRollout(sessionsDir: string): string | null {
|
||||
if (!existsSync(sessionsDir)) return null;
|
||||
const found: Array<{ path: string; mtime: number }> = [];
|
||||
const walk = (dir: string) => {
|
||||
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
||||
const full = join(dir, entry.name);
|
||||
if (entry.isDirectory()) walk(full);
|
||||
else if (entry.name.startsWith('rollout-') && entry.name.endsWith('.jsonl')) {
|
||||
found.push({ path: full, mtime: statSync(full).mtimeMs });
|
||||
}
|
||||
}
|
||||
};
|
||||
walk(sessionsDir);
|
||||
if (found.length === 0) return null;
|
||||
found.sort((a, b) => b.mtime - a.mtime);
|
||||
return found[0].path;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
|
||||
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
|
||||
* distinct tasks sharing a truncated prefix (e.g. `foo--hash` and `foo--hash-2`,
|
||||
* both truncating to `foo--ha`) collide: a dirname-prefix scan returns whichever
|
||||
* sorts first and misroutes the other's trials (observed in the wild as base +
|
||||
* `-2` reference-runs sharing trial IDs). result.json is written per-trial with
|
||||
* the real task_name, so it disambiguates exactly. Returns null when result.json
|
||||
* is absent/unparseable or names a task dir that doesn't exist (caller then falls
|
||||
* back to the prefix scan).
|
||||
*/
|
||||
function findTaskDirByResultJson(trialPath: string): string | null {
|
||||
const resultPath = join(trialPath, 'result.json');
|
||||
if (!existsSync(resultPath)) return null;
|
||||
let taskName: unknown;
|
||||
try {
|
||||
taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown }).task_name;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (typeof taskName !== 'string' || taskName.length === 0) return null;
|
||||
// Hub-published task_names are org-prefixed (`<org>/<slug>`); the dir is bare.
|
||||
// Inlined (not the shared bareSlug helper) because this script ships in the
|
||||
// worker toolkit and must not import outside its shipped file set.
|
||||
const dir = join(HARBOR_TASKS_DIR, taskName.replace(/^[^/]+\//, ''));
|
||||
return existsSync(dir) ? dir : null;
|
||||
}
|
||||
|
||||
function copyTrial(trialPath: string, destName?: string) {
|
||||
trialPath = trialPath.replace(/\/$/, '');
|
||||
|
||||
if (!existsSync(trialPath)) {
|
||||
console.error(`Error: ${trialPath} does not exist`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Repair the SOURCE before reading a byte of it. A trial can leave files
|
||||
// write-only, which locks out their own owner: everything below — reading
|
||||
// reward.txt, copying agent-output — fails on them, and any that do get
|
||||
// through land in the task dir, where harbor hashes every file on every
|
||||
// later trial and one unreadable path aborts the run.
|
||||
let sourcePerms = null;
|
||||
try {
|
||||
sourcePerms = normalizeTreePermissions(trialPath);
|
||||
} catch (err) {
|
||||
console.warn(`Warning: could not normalize permissions on ${trialPath}: ${String(err)}`);
|
||||
console.warn(` If the copy below fails on permissions:`);
|
||||
console.warn(` ${manualRepairHint(trialPath)}`);
|
||||
}
|
||||
if (sourcePerms && sourcePerms.failures.length > 0) {
|
||||
console.warn(
|
||||
`Warning: could not normalize permissions on ${sourcePerms.failures.length} path(s) under ${trialPath}.`
|
||||
);
|
||||
console.warn(` If the copy below fails on permissions, run:`);
|
||||
console.warn(` ${manualRepairHint(trialPath)}`);
|
||||
}
|
||||
|
||||
const rewardPath = join(trialPath, 'verifier', 'reward.txt');
|
||||
if (!existsSync(rewardPath)) {
|
||||
console.error(`Error: No reward.txt found in ${trialPath}/verifier/`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const reward = readFileSync(rewardPath, 'utf-8').trim();
|
||||
const trialDir = basename(trialPath);
|
||||
|
||||
// Trial dir format: <task-slug-truncated>__<trialId>
|
||||
const separatorIndex = trialDir.lastIndexOf('__');
|
||||
if (separatorIndex === -1) {
|
||||
console.error(
|
||||
`Error: Trial directory '${trialDir}' does not match expected format <slug>__<trialId>`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const trialPrefix = trialDir.substring(0, separatorIndex);
|
||||
const trialId = trialDir.substring(separatorIndex + 2);
|
||||
|
||||
// Prefer the exact task_name from result.json (handles truncated-prefix
|
||||
// collisions like `foo--hash` vs `foo--hash-2`); fall back to the dirname
|
||||
// prefix scan only when result.json can't resolve it.
|
||||
const taskDir = findTaskDirByResultJson(trialPath) ?? findTaskDir(trialPrefix);
|
||||
if (!taskDir) {
|
||||
console.error(
|
||||
`Error: Could not find task directory matching prefix '${trialPrefix}' in ${HARBOR_TASKS_DIR}/`
|
||||
);
|
||||
console.error('Available tasks:');
|
||||
readdirSync(HARBOR_TASKS_DIR).forEach((d) => console.error(` ${d}`));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// destName (--dest-name) makes a RECORDED rollout id authoritative: the run
|
||||
// is copied to exactly that name instead of the minted reward-<r>-<id> —
|
||||
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
|
||||
// the publish-manifest run_id stay byte-identical by construction.
|
||||
const dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
|
||||
|
||||
if (existsSync(dest)) {
|
||||
console.warn(`Warning: ${dest} already exists, overwriting`);
|
||||
rmSync(dest, { recursive: true });
|
||||
}
|
||||
|
||||
mkdirSync(dest, { recursive: true });
|
||||
|
||||
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
|
||||
// companion reward-correctness.txt (N/A by design) and the machine-readable
|
||||
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
|
||||
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
|
||||
// (the source of truth grade.md/reward.txt are rendered from), the
|
||||
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
|
||||
// render-stderr(-<N>).log, and grader-samples.txt.
|
||||
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
|
||||
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
|
||||
// every per-sample record.)
|
||||
//
|
||||
// signals-status.txt qualifies the deterministic signals the grader was fed:
|
||||
// it records whether every deterministic check actually produced a verdict
|
||||
// ("ok") or one or more was killed before finishing ("degraded" — the grade
|
||||
// is then NOT fully signal-backed). Without it a copied run is
|
||||
// indistinguishable from a run whose checks all passed, so it must travel
|
||||
// with the reward files.
|
||||
const verifierDir = join(trialPath, 'verifier');
|
||||
if (existsSync(verifierDir)) {
|
||||
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
|
||||
if (!entry.isFile()) continue;
|
||||
const f = entry.name;
|
||||
if (
|
||||
f === 'reward.txt' ||
|
||||
f === 'reward-correctness.txt' ||
|
||||
f === 'reward.json' ||
|
||||
f === 'signals-status.txt' ||
|
||||
f === 'grader-samples.txt' ||
|
||||
f === 'grader-regime.json' ||
|
||||
/^grade(-\d+)?\.md$/.test(f) ||
|
||||
/^grade(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-result(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
|
||||
/^render-stderr(-\d+)?\.log$/.test(f)
|
||||
) {
|
||||
copyPath(join(verifierDir, f), join(dest, f));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const agentOutputDir = join(verifierDir, 'agent-output');
|
||||
if (existsSync(agentOutputDir)) {
|
||||
copyTree(agentOutputDir, join(dest, 'agent-output'));
|
||||
}
|
||||
|
||||
// Copy agent session log and trajectory (not workspace)
|
||||
const agentDir = join(trialPath, 'agent');
|
||||
if (existsSync(agentDir)) {
|
||||
mkdirSync(join(dest, 'agent'), { recursive: true });
|
||||
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
|
||||
// the Claude one left codex reference runs with nothing but the trajectory.
|
||||
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
|
||||
const src = join(agentDir, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, 'agent', file));
|
||||
}
|
||||
// The grader (and downstream worldbench export / replay) reads
|
||||
// agent/trajectory.json. If it's missing, scream so we don't silently
|
||||
// ship a reference run that's only half-useful.
|
||||
if (!existsSync(join(agentDir, 'trajectory.json'))) {
|
||||
console.warn(
|
||||
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
|
||||
`Harbor's adapter for this harness failed to write it (typically because ` +
|
||||
`the converter choked on the session log). Downstream consumers ` +
|
||||
`(grader replay, worldbench export) need this file — investigate ` +
|
||||
`before relying on this reference run.`
|
||||
);
|
||||
}
|
||||
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
|
||||
// gets embedded in claude-code.txt's first non-empty line; we use it to
|
||||
// locate the sibling JSONL Claude Code wrote in the same trial.
|
||||
// Layout is per harness, so each needs a case here — the same reason
|
||||
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
|
||||
// session, which is what codex got before this: nothing at all.
|
||||
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
|
||||
if (existsSync(claudeCodeTxt)) {
|
||||
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
|
||||
const sessionId = readSessionId(claudeCodeTxt);
|
||||
if (sessionId) {
|
||||
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
|
||||
if (existsSync(jsonl)) copyPath(jsonl, join(dest, 'session.jsonl'));
|
||||
}
|
||||
} else {
|
||||
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
|
||||
const rollout = newestRollout(join(agentDir, 'sessions'));
|
||||
if (rollout) copyPath(rollout, join(dest, 'session.jsonl'));
|
||||
}
|
||||
}
|
||||
|
||||
// Copy top-level metadata
|
||||
for (const file of ['config.json', 'result.json', 'trial.log']) {
|
||||
const src = join(trialPath, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, file));
|
||||
}
|
||||
|
||||
// Record the checksums of the task inputs this run was generated against
|
||||
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
|
||||
// submit-task.ts re-captures at packaging time and warns when any of them
|
||||
// changed — the run then describes an older revision of the task than the
|
||||
// one being shipped. Preferred source: the launch-time stamp harbor-run
|
||||
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
|
||||
// 'run') — it records the inputs the agent actually ran against, so an
|
||||
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
|
||||
// (trials from an older harbor-run, or a failed stamp): capture here at
|
||||
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
|
||||
// the weaker evidence.
|
||||
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
|
||||
const trialStamp = readTaskInputChecksums(trialStampPath);
|
||||
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
|
||||
// harbor-runs sharing a cwd) — its hashes describe some other task's
|
||||
// inputs, so treat it as absent rather than importing false evidence.
|
||||
const stampSlug = trialStamp?.taskSlug;
|
||||
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
|
||||
if (trialStamp && !stampMisrouted) {
|
||||
copyPath(trialStampPath, join(dest, INPUT_CHECKSUMS_FILENAME));
|
||||
} else {
|
||||
if (stampMisrouted) {
|
||||
console.warn(
|
||||
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
|
||||
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
|
||||
);
|
||||
}
|
||||
writeFileSync(
|
||||
join(dest, INPUT_CHECKSUMS_FILENAME),
|
||||
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
|
||||
);
|
||||
}
|
||||
|
||||
// Files captured from a run can land unreadable to you, which makes packaging
|
||||
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
|
||||
let perms = null;
|
||||
try {
|
||||
perms = normalizeTreePermissions(dest);
|
||||
} catch (err) {
|
||||
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
|
||||
console.warn(` The run copied fine. If packaging later fails on permissions:`);
|
||||
console.warn(` ${manualRepairHint(dest)}`);
|
||||
}
|
||||
if (perms && perms.failures.length > 0) {
|
||||
console.warn(
|
||||
`Warning: could not normalize permissions on ${perms.failures.length} path(s) under ${dest}.`
|
||||
);
|
||||
console.warn(
|
||||
` If packaging later fails with 'Cannot stat: Permission denied', run:\n` +
|
||||
` ${manualRepairHint(dest)}`
|
||||
);
|
||||
}
|
||||
|
||||
console.log(`Copied to ${dest}`);
|
||||
console.log(` reward: ${reward}`);
|
||||
console.log(` task: ${taskDir}`);
|
||||
console.log(` trial: ${trialId}`);
|
||||
const ownerFixed = (sourcePerms?.ownerFixed.length ?? 0) + (perms?.ownerFixed.length ?? 0);
|
||||
const modeFixed = (sourcePerms?.modeFixed.length ?? 0) + (perms?.modeFixed.length ?? 0);
|
||||
if (ownerFixed > 0 || modeFixed > 0) {
|
||||
console.log(` perms: normalized ${ownerFixed} owner / ${modeFixed} mode`);
|
||||
}
|
||||
}
|
||||
|
||||
// Main
|
||||
const rawArgs = process.argv.slice(2);
|
||||
let destName: string | undefined;
|
||||
const args: string[] = [];
|
||||
for (let i = 0; i < rawArgs.length; i++) {
|
||||
if (rawArgs[i] === '--dest-name') {
|
||||
destName = rawArgs[++i];
|
||||
if (!destName) {
|
||||
console.error('Error: --dest-name requires a value');
|
||||
process.exit(1);
|
||||
}
|
||||
} else {
|
||||
args.push(rawArgs[i]);
|
||||
}
|
||||
}
|
||||
|
||||
if (args.length === 0) {
|
||||
console.error(
|
||||
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
if (destName && args.length !== 1) {
|
||||
console.error('Error: --dest-name applies to exactly one trial path');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
for (const trialPath of args) {
|
||||
copyTrial(trialPath, destName);
|
||||
}
|
||||
98
worker-toolkit-potion-polyglot-orig/scripts/dnsjail.py
Normal file
98
worker-toolkit-potion-polyglot-orig/scripts/dnsjail.py
Normal file
@@ -0,0 +1,98 @@
|
||||
"""Apply the DNS jail to a trial container: the model endpoint resolves, nothing else does.
|
||||
|
||||
Opt-in with RACCOON_DNS_JAIL=1. Runs from the agent's own turn rather than from a compose
|
||||
overlay — the allowlist comes from the proxy URL this process already holds (plus any hosts
|
||||
RACCOON_DNS_JAIL_ALLOW adds), so nothing has to be injected into the container, and the jail works on every harbor backend. Deliberately
|
||||
after agent-setup: a harness that downloads its CLI there still reaches the network to do it.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
from typing import Any
|
||||
|
||||
JAIL = "/usr/local/bin/raccoon-dns-jail"
|
||||
_NO_SCRIPT = "raccoon-dns-jail: not in this image"
|
||||
|
||||
_URL_VARS = (
|
||||
"ANTHROPIC_BASE_URL", "OPENAI_BASE_URL", "GOOGLE_GEMINI_BASE_URL",
|
||||
"HTTPS_PROXY", "https_proxy", "HTTP_PROXY", "http_proxy", "ALL_PROXY", "all_proxy",
|
||||
)
|
||||
|
||||
_EXTRA_VAR = "RACCOON_DNS_JAIL_ALLOW"
|
||||
|
||||
_log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _host(url: str) -> str:
|
||||
"""Hostname out of a URL, or "" when it is not a plain hostname we can allow."""
|
||||
h = url.split("://", 1)[-1].split("/", 1)[0].rsplit("@", 1)[-1].split(":", 1)[0]
|
||||
if not h or h.startswith((".", "-")) or h.endswith(".") or not all(
|
||||
c.isascii() and (c.isalnum() or c in ".-") for c in h
|
||||
):
|
||||
return ""
|
||||
# An IP-literal endpoint (a loopback proxy shim, say) needs no DNS at all, and a
|
||||
# --server rule for it would only be checked by a PTR query the catch-all answers.
|
||||
if all(part.isdigit() for part in h.split(".")):
|
||||
return ""
|
||||
return h
|
||||
|
||||
|
||||
def dns_jail_allowlist() -> tuple[list[str], list[str]]:
|
||||
"""(required, advisory).
|
||||
|
||||
Required = the hosts this process's own env says the agent will dial; every one must
|
||||
resolve through the jail or no jail is applied, because a host the agent needs and
|
||||
cannot resolve is a dead trial. Advisory = whatever RACCOON_DNS_JAIL_ALLOW adds, which
|
||||
only warns: an added host that CNAMEs outside the allowlist cannot resolve through the
|
||||
catch-all, and must not take the whole jail down with it.
|
||||
"""
|
||||
required: list[str] = []
|
||||
for var in _URL_VARS:
|
||||
h = _host(os.environ.get(var) or "")
|
||||
if h and h not in required:
|
||||
required.append(h)
|
||||
advisory: list[str] = []
|
||||
for entry in (os.environ.get(_EXTRA_VAR) or "").replace(",", " ").split():
|
||||
# Bare hostnames only: a URL silently truncated to its first path segment would
|
||||
# allow a name nobody asked for and block the one they meant.
|
||||
h = "" if ("/" in entry or ":" in entry) else _host(entry)
|
||||
if not h:
|
||||
_log.warning("DNS jail: ignoring unusable %s entry %r", _EXTRA_VAR, entry)
|
||||
elif h not in required and h not in advisory:
|
||||
advisory.append(h)
|
||||
return required, advisory
|
||||
|
||||
|
||||
def dns_jail_enabled() -> bool:
|
||||
return os.environ.get("RACCOON_DNS_JAIL") == "1"
|
||||
|
||||
|
||||
async def apply_dns_jail(agent: Any, environment: Any) -> None:
|
||||
"""No-op unless enabled; leaves the container's DNS untouched on any doubt."""
|
||||
if not dns_jail_enabled():
|
||||
return
|
||||
required, advisory = dns_jail_allowlist()
|
||||
allow = " ".join(required)
|
||||
# A blank allowlist means no model endpoint was found: jailing would strand the agent.
|
||||
if not allow:
|
||||
_log.warning("DNS jail: no usable model endpoint — the trial keeps normal network access")
|
||||
return
|
||||
try:
|
||||
result = await agent.exec_as_root(
|
||||
environment,
|
||||
command=(
|
||||
f"if [ -x {JAIL} ]; then DNSJAIL_ALLOW={shlex.quote(allow)} "
|
||||
f"DNSJAIL_ALLOW_EXTRA={shlex.quote(' '.join(advisory))} {JAIL}; "
|
||||
f'else echo "{_NO_SCRIPT}"; fi'
|
||||
),
|
||||
)
|
||||
except Exception as exc: # a jail that cannot be applied must not fail the trial
|
||||
_log.warning("DNS jail: could not apply (%s) — the trial keeps normal network access", exc)
|
||||
return
|
||||
# An image frozen before this feature has nothing to invoke. Say so: a launcher that
|
||||
# believes the network is restricted when it is not is worse than no jail at all.
|
||||
if _NO_SCRIPT in (getattr(result, "stdout", "") or ""):
|
||||
_log.warning(
|
||||
"DNS jail: this task's image ships no resolver — the trial keeps normal network access"
|
||||
)
|
||||
48
worker-toolkit-potion-polyglot-orig/scripts/guidance-target.sh
Executable file
48
worker-toolkit-potion-polyglot-orig/scripts/guidance-target.sh
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/bin/bash
|
||||
# guidance-target.sh — print the holistic-rubric file the grader reads for a task.
|
||||
#
|
||||
# The grader reads the task's holistic rubric under the Grading Standard.
|
||||
# Detector skills call this resolver so they always assess the file the grader
|
||||
# will actually read, and so the resolution rule lives in one place.
|
||||
#
|
||||
# Resolution order (renames are forward-only, so every generation stays readable):
|
||||
# tests/holistic-rubric.md the current name; new tasks use it
|
||||
# tests/grader-guidance-consolidated.md tasks created before the rename
|
||||
# tests/grader-guidance.md legacy-generation tasks
|
||||
# When none exists yet, the current name is printed — that is the file a new
|
||||
# task's rubric will be written to.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/guidance-target.sh <slug-or-task-dir>
|
||||
#
|
||||
# Output (one line): the path to the rubric file.
|
||||
set -eu
|
||||
|
||||
arg="${1:?usage: bash scripts/guidance-target.sh <slug-or-task-dir>}"
|
||||
dir="$arg"
|
||||
[ -d "$dir" ] || dir="harbor-tasks/$arg"
|
||||
tests="$dir/tests"
|
||||
[ -d "$tests" ] || { echo "ERROR: no tests/ directory at $dir" >&2; exit 1; }
|
||||
|
||||
new="$tests/holistic-rubric.md"
|
||||
old="$tests/grader-guidance-consolidated.md"
|
||||
legacy="$tests/grader-guidance.md"
|
||||
|
||||
if [ -f "$new" ] && [ -f "$old" ]; then
|
||||
# Both names present: the grader's pick depends on the harness generation,
|
||||
# so an assessment of either could be an assessment of the wrong file.
|
||||
# Byte-identical copies are safe; anything else is a hard stop.
|
||||
if ! cmp -s "$new" "$old"; then
|
||||
echo "ERROR: $tests carries both holistic-rubric.md and grader-guidance-consolidated.md with different content — keep exactly one (tests/holistic-rubric.md is the current name)" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "$new"
|
||||
elif [ -f "$new" ]; then
|
||||
echo "$new"
|
||||
elif [ -f "$old" ]; then
|
||||
echo "$old"
|
||||
elif [ -f "$legacy" ]; then
|
||||
echo "$legacy"
|
||||
else
|
||||
echo "$new"
|
||||
fi
|
||||
299
worker-toolkit-potion-polyglot-orig/scripts/harbor-regrade
Executable file
299
worker-toolkit-potion-polyglot-orig/scripts/harbor-regrade
Executable file
@@ -0,0 +1,299 @@
|
||||
#!/bin/bash
|
||||
# Re-grade an existing reference run without re-invoking the agent.
|
||||
#
|
||||
# Spins up a normal harbor trial, but plugs in scripts/replay_agent.py
|
||||
# instead of a real agent. The replay agent overlays the captured
|
||||
# agent-output into /workspace, applies any captured deletions, drops
|
||||
# the captured trajectory at /logs/agent/trajectory.json so the grader
|
||||
# reads the same transcript it would for the original run, then exits.
|
||||
# The verifier (real test.sh, real LLM grader if present) runs as it
|
||||
# would for any other trial.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
#
|
||||
# Examples:
|
||||
# # Single regrade
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
|
||||
#
|
||||
# # Ten regrades of the same reference run (independent grader trials)
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg \
|
||||
# -k 10
|
||||
#
|
||||
# --fast runs the GRADER in claude's fast serving mode (faster output at a higher
|
||||
# token rate). The replay agent runs no model, so the grader is the only model in
|
||||
# this path. Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# See scripts/replay_agent.py for what the agent actually does, and the
|
||||
# `verifier: capture tracked-file deletions in agent-output` PR for the
|
||||
# capture half of this flow (_HARBOR_DELETIONS.txt).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# Source API key + any verifier env from the repo's .env
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
source "$REPO_ROOT/.env"
|
||||
set +a
|
||||
fi
|
||||
|
||||
# The grader authenticates with ANTHROPIC_API_KEY straight out of the .env sourced above, so
|
||||
# a .env saved on Windows would hand it a value with a carriage return still attached.
|
||||
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
|
||||
# shellcheck disable=SC1091
|
||||
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
|
||||
harness_setup_credentials >/dev/null 2>&1 || true
|
||||
fi
|
||||
|
||||
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
|
||||
fi
|
||||
|
||||
# When running inside a devcontainer, harbor needs HOST paths for docker
|
||||
# bind mounts (the docker daemon is on the host).
|
||||
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
|
||||
cd "$HOST_WORKSPACE"
|
||||
fi
|
||||
|
||||
usage() {
|
||||
cat >&2 <<EOF
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
|
||||
Required arguments:
|
||||
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
|
||||
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
|
||||
agent-output/ (and ideally agent/trajectory.json).
|
||||
|
||||
Optional arguments:
|
||||
--fast grade in claude's fast serving mode (higher token rate,
|
||||
faster output). Anything else is passed through to harbor.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
|
||||
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
|
||||
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
|
||||
FAST_REQUESTED=""
|
||||
REGRADE_ARGS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--fast) FAST_REQUESTED=1; shift ;;
|
||||
*) REGRADE_ARGS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
|
||||
|
||||
[ $# -lt 2 ] && usage
|
||||
TASK_DIR="$1"
|
||||
REF_RUN_DIR="$2"
|
||||
shift 2
|
||||
|
||||
# Resolve to absolute paths — harbor cd's around internally; the replay
|
||||
# agent receives the path as an --agent-kwarg and won't know our cwd.
|
||||
TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: task-dir does not exist: $TASK_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Newer-generation Dockerfiles COPY environment/dns-jail/, a derived directory
|
||||
# that harbor-run pre-stages but a bare regrade context may lack — the sandbox
|
||||
# build then fails before the verifier ever starts. Recreate it the same way.
|
||||
if [ -d "$TASK_DIR_ABS/environment" ]; then
|
||||
mkdir -p "$TASK_DIR_ABS/environment/dns-jail"
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Rubric grader modes (HARBOR_GRADER_MODE=rubric-*) read tests/render-rubric-grade.py,
|
||||
# a shared asset like the dns-jail script above. A task created before the rubric
|
||||
# renderer shipped has no copy, and a stale copy aggregates with outdated weights;
|
||||
# either way the regrade must run against the current shared copy. Stage it
|
||||
# host-side — the container only ever sees the task directory. The source lives at
|
||||
# harbor-tasks/raccoon-shared/ in the internal repo and task-shared/ in a worker
|
||||
# toolkit checkout; first one present wins.
|
||||
case "${HARBOR_GRADER_MODE:-}" in
|
||||
rubric-*)
|
||||
RUBRIC_RENDER_DEST="$TASK_DIR_ABS/tests/render-rubric-grade.py"
|
||||
for RUBRIC_RENDER_SRC in "$REPO_ROOT/harbor-tasks/raccoon-shared/render-rubric-grade.py" \
|
||||
"$REPO_ROOT/task-shared/render-rubric-grade.py"; do
|
||||
[ -f "$RUBRIC_RENDER_SRC" ] || continue
|
||||
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
|
||||
mkdir -p "$TASK_DIR_ABS/tests"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
|
||||
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
|
||||
fi
|
||||
break
|
||||
done
|
||||
;;
|
||||
esac
|
||||
|
||||
# NOTE: agent-output/ is intentionally NOT required here. Advisory tasks (the
|
||||
# agent only reads + answers in chat) make no workspace edits, so a faithful
|
||||
# capture has an empty/absent agent-output/ — the deliverable lives in the
|
||||
# captured transcript (agent/trajectory.json) that the grader reads. ReplayAgent
|
||||
# overlays agent-output/ when present and otherwise grades base-workspace +
|
||||
# transcript, but FAILS LOUDLY if the transcript shows file-mutating tool calls
|
||||
# with no agent-output/ (genuine lost edits). So we let it make that call.
|
||||
if [ ! -d "$REF_RUN_DIR_ABS/agent-output" ]; then
|
||||
echo "Note: $REF_RUN_DIR_ABS has no agent-output/ — replaying as an" >&2
|
||||
echo " advisory run (base workspace + captured transcript). See" >&2
|
||||
echo " scripts/replay_agent.py for the lost-edits safety guard." >&2
|
||||
fi
|
||||
|
||||
# Make scripts/ importable so harbor can find replay_agent:ReplayAgent.
|
||||
# ${PYTHONPATH:+...} so an unset PYTHONPATH doesn't leave a trailing colon —
|
||||
# python treats the resulting empty entry as the CWD, silently putting
|
||||
# whatever directory the user ran this from on harbor's sys.path.
|
||||
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
|
||||
|
||||
# Environment backend. Explicit HARBOR_ENV wins; otherwise default to docker in
|
||||
# a worker-toolkit checkout (detected by toolkit.json at the repo root) and
|
||||
# daytona in the internal repo. See scripts/harbor-run for the full rationale
|
||||
# (why the toolkit needs docker, why the marker is a workspace file not an image
|
||||
# env, and why daytona must NOT pass --no-delete — billed sandbox).
|
||||
if [ -n "${HARBOR_ENV:-}" ]; then
|
||||
ENV_TYPE="$HARBOR_ENV"
|
||||
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
|
||||
ENV_TYPE="docker"
|
||||
else
|
||||
ENV_TYPE="daytona"
|
||||
fi
|
||||
DELETE_FLAGS="--no-delete"
|
||||
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
|
||||
|
||||
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
|
||||
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
|
||||
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
|
||||
RESOURCE_FLAGS=""
|
||||
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
|
||||
|
||||
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
|
||||
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
|
||||
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Output dir. harbor names the job subdir by second-granularity timestamp, so
|
||||
# many regrades launched in the same second under one -o collide
|
||||
# ("Job directory ... already exists and cannot be resumed"). Set
|
||||
# HARBOR_REGRADE_OUT to a per-run unique dir when running a parallel sweep.
|
||||
OUT_DIR="${HARBOR_REGRADE_OUT:-harbor-jobs}"
|
||||
|
||||
# Optional grader mode: HARBOR_GRADER_MODE=one-shot flips the task's test.sh into
|
||||
# the no-tools one-shot grader (vs the default agentic grader) via verifier env —
|
||||
# lets us A/B the agenticity gap without forking the task. See raccoon-shared/test.sh.
|
||||
GRADER_MODE_FLAG=()
|
||||
[ -n "${HARBOR_GRADER_MODE:-}" ] && GRADER_MODE_FLAG=(--verifier-env "GRADER_MODE=$HARBOR_GRADER_MODE")
|
||||
|
||||
# Optional grader model: HARBOR_GRADER_MODEL=claude-fable-5-1 overrides the grader's
|
||||
# model (default: the `opus` alias) via verifier env — lets us A/B the grader model
|
||||
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
|
||||
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
|
||||
|
||||
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
|
||||
# For measuring per-sample properties of the grader (e.g. how often it emits a
|
||||
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
|
||||
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
|
||||
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
|
||||
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
|
||||
|
||||
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
|
||||
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
|
||||
# env var Claude Code itself reads, so test.sh needs no knowledge of it). For
|
||||
# proxies whose responses outlast the CLI default.
|
||||
[ -n "${HARBOR_API_TIMEOUT_MS:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "API_TIMEOUT_MS=$HARBOR_API_TIMEOUT_MS")
|
||||
|
||||
# Fast serving mode for the grader's own claude calls (--fast). Per-task tests/ assets
|
||||
# are frozen at creation, so say when this task's copy cannot act on the flag. The note
|
||||
# names that copy, not a shared path — this script ships to workers under another layout.
|
||||
if [ -n "$FAST_REQUESTED" ]; then
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_FAST_MODE=true")
|
||||
if [ -f "$TASK_DIR_ABS/tests/test.sh" ] &&
|
||||
! grep -q 'GRADER_FAST_MODE' "$TASK_DIR_ABS/tests/test.sh"; then
|
||||
echo "Note: --fast passed, but this task's tests/test.sh does not read" >&2
|
||||
echo " GRADER_FAST_MODE, so the grader will run at normal speed —" >&2
|
||||
echo " its copy predates the flag. Refresh the task's tests/test.sh" >&2
|
||||
echo " from the current shared grader assets to enable it." >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# Carry the SOURCE run's agent identity + model into this replay's own record.
|
||||
#
|
||||
# A replay reports `replay_agent:ReplayAgent` with model_name null, because no model
|
||||
# ran — the behaviour being graded came from the source run. Recording only
|
||||
# reference_run_dir makes that a pointer, and pointers dangle: a regrade is normally
|
||||
# copied back over the run it regraded, so the source usually no longer exists (501 of
|
||||
# 643 on-disk replays already point at a missing dir, none of them in the published
|
||||
# manifest either). Stamping the values here makes the replay self-describing, so the
|
||||
# originating harness and model survive the source's deletion.
|
||||
#
|
||||
# Regrading a REGRADE means the source is itself a replay, so copying its own identity
|
||||
# forward would overwrite the real provenance with a self-reference: inherit what it
|
||||
# inherited instead.
|
||||
#
|
||||
# Read with python3 rather than jq — jq is not guaranteed on a worker's box, and a
|
||||
# missing source result.json must degrade to "unknown", never abort the regrade.
|
||||
SOURCE_PROV_FLAGS=()
|
||||
if [ -f "$REF_RUN_DIR_ABS/result.json" ]; then
|
||||
SOURCE_PROV=$(python3 -c '
|
||||
import json, sys
|
||||
try:
|
||||
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
|
||||
except Exception:
|
||||
sys.exit(0)
|
||||
kw = a.get("kwargs") or {}
|
||||
REPLAY = "replay_agent:ReplayAgent"
|
||||
if (a.get("import_path") or a.get("name")) == REPLAY:
|
||||
agent, model = kw.get("source_agent_import_path"), kw.get("source_model_name")
|
||||
else:
|
||||
agent, model = a.get("import_path") or a.get("name"), a.get("model_name")
|
||||
# An already-damaged chain cannot be recovered; leave it honestly unstamped rather
|
||||
# than propagating a source that names the replay agent itself.
|
||||
print("" if agent == REPLAY else agent or "")
|
||||
print(model or "")
|
||||
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
|
||||
SOURCE_AGENT=$(printf '%s\n' "$SOURCE_PROV" | sed -n 1p)
|
||||
SOURCE_MODEL=$(printf '%s\n' "$SOURCE_PROV" | sed -n 2p)
|
||||
[ -n "$SOURCE_AGENT" ] && SOURCE_PROV_FLAGS+=(--ak "source_agent_import_path=$SOURCE_AGENT")
|
||||
[ -n "$SOURCE_MODEL" ] && SOURCE_PROV_FLAGS+=(--ak "source_model_name=$SOURCE_MODEL")
|
||||
fi
|
||||
|
||||
# HARBOR_GRADING_STANDARD, when set, is passed through to the task's own
|
||||
# tests/test.sh as the GRADING_STANDARD verifier env var. What (if anything)
|
||||
# it does is decided by the scripts inside that tests/ directory; a test.sh
|
||||
# that reads no such variable ignores it.
|
||||
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
|
||||
|
||||
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
|
||||
# nothing to restrict — only the verifier runs, and it needs the proxy.
|
||||
exec harbor run \
|
||||
-p "$TASK_DIR_ABS" \
|
||||
--agent-import-path replay_agent:ReplayAgent \
|
||||
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
|
||||
${SOURCE_PROV_FLAGS[@]+"${SOURCE_PROV_FLAGS[@]}"} \
|
||||
-e "$ENV_TYPE" \
|
||||
$DELETE_FLAGS \
|
||||
$RESOURCE_FLAGS \
|
||||
--yes \
|
||||
-o "$OUT_DIR" \
|
||||
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
|
||||
"$@"
|
||||
467
worker-toolkit-potion-polyglot-orig/scripts/harbor-run
Executable file
467
worker-toolkit-potion-polyglot-orig/scripts/harbor-run
Executable file
@@ -0,0 +1,467 @@
|
||||
#!/bin/bash
|
||||
# Run raccoon tasks via Harbor with standard defaults.
|
||||
#
|
||||
# Automatically detects snapshot-based tasks (those with environment/session.jsonl)
|
||||
# and uses the snapshot agent adapter for session resume.
|
||||
#
|
||||
# Usage: scripts/harbor-run <task-dir> [extra harbor args...]
|
||||
# Example: scripts/harbor-run harbor-tasks/my-task-slug
|
||||
# Example: scripts/harbor-run harbor-tasks/my-task-slug -k 4 --force-build
|
||||
#
|
||||
# To change the model, use --model (consumed here). Passing harbor's own -m does NOT
|
||||
# override: harbor's -m is repeatable and builds one agent per value, so `-m X` runs the
|
||||
# registry default AND X — two trials.
|
||||
#
|
||||
# --fast runs both models of the trial in fast serving mode (faster output at a higher
|
||||
# token rate): the trial agent (claude-code only) and the grader the verifier launches.
|
||||
# Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
|
||||
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# An explicit RACCOON_DNS_JAIL from the caller wins over .env, so a worker who opted in
|
||||
# there can still turn the jail off for one run.
|
||||
_RJ_SET="${RACCOON_DNS_JAIL+set}"; _RJ_VAL="${RACCOON_DNS_JAIL:-}"
|
||||
_RJA_SET="${RACCOON_DNS_JAIL_ALLOW+set}"; _RJA_VAL="${RACCOON_DNS_JAIL_ALLOW:-}"
|
||||
|
||||
# Source API key
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
source "$REPO_ROOT/.env"
|
||||
set +a
|
||||
fi
|
||||
|
||||
if [ -n "$_RJ_SET" ]; then RACCOON_DNS_JAIL="$_RJ_VAL"; fi
|
||||
if [ -n "$_RJA_SET" ]; then RACCOON_DNS_JAIL_ALLOW="$_RJA_VAL"; fi
|
||||
|
||||
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
|
||||
fi
|
||||
|
||||
# Per-harness credentials (OPENAI_API_KEY and friends) are DERIVED from the proxy root in
|
||||
# ANTHROPIC_BASE_URL — they are not in .env. The container-create derivation exported them
|
||||
# into a process that has long since exited, and only the auth FILES it wrote survive, so a
|
||||
# fresh shell has the key on disk but not in its environment. resolve_harness checks the
|
||||
# environment, and the trial passes it through to the sandbox, so re-derive here.
|
||||
# Quiet on purpose: if it does not work, resolve_harness refuses by name a second later.
|
||||
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
|
||||
# shellcheck disable=SC1091
|
||||
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
|
||||
harness_setup_credentials >/dev/null 2>&1 || true
|
||||
fi
|
||||
|
||||
export ANTHROPIC_BASE_URL="${ANTHROPIC_BASE_URL:-}"
|
||||
|
||||
# When running inside a devcontainer, harbor computes absolute paths for
|
||||
# Docker bind mounts. These paths must be HOST paths because docker compose
|
||||
# talks to the host daemon via the shared socket. Switching CWD to the
|
||||
# host-equivalent workspace path makes harbor resolve paths correctly.
|
||||
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
|
||||
cd "$HOST_WORKSPACE"
|
||||
fi
|
||||
|
||||
TASK_DIR="$1"
|
||||
shift
|
||||
|
||||
# Harness selection. `--harness` / `--model` / `--fast` / `--check-model` are consumed
|
||||
# here; everything else passes through to harbor untouched, so existing invocations keep
|
||||
# working. Parsed with a loop rather than getopts because the remaining args are an
|
||||
# opaque harbor passthrough that getopts would try to interpret.
|
||||
HARNESS_ARGS=()
|
||||
PASSTHROUGH=()
|
||||
FAST_REQUESTED=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--harness) HARNESS_ARGS+=(--harness "$2"); shift 2 ;;
|
||||
--harness=*) HARNESS_ARGS+=(--harness "${1#*=}"); shift ;;
|
||||
--model) HARNESS_ARGS+=(--model "$2"); shift 2 ;;
|
||||
--model=*) HARNESS_ARGS+=(--model "${1#*=}"); shift ;;
|
||||
--fast) HARNESS_ARGS+=(--fast); FAST_REQUESTED=1; shift ;;
|
||||
--check-model) HARNESS_ARGS+=(--check-model); shift ;;
|
||||
*) PASSTHROUGH+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${PASSTHROUGH[@]+"${PASSTHROUGH[@]}"}"
|
||||
|
||||
# Preflight: workspace must be populated before harbor tries to docker-build it.
|
||||
# Without this, the Dockerfile's `COPY workspace/ .` fails with an opaque
|
||||
# "failed to calculate checksum of ref ...: \"/workspace\": not found" buried
|
||||
# several frames deep in harbor's asyncio + docker-compose traceback. Surface
|
||||
# the real fix here instead.
|
||||
WORKSPACE_DIR="$TASK_DIR/environment/workspace"
|
||||
if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ]; then
|
||||
SLUG="$(basename "$TASK_DIR")"
|
||||
echo "Error: $WORKSPACE_DIR is missing or empty." >&2
|
||||
echo "Build it first: bash scripts/build-workspace.sh $SLUG" >&2
|
||||
echo "(reads commit from $TASK_DIR/task.toml; applies environment/workspace.patch if present.)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Preflight: refuse a task dir harbor would not accept as a task.
|
||||
#
|
||||
# `harbor run -p` falls back to reading a rejected dir as a DATASET of tasks, finds none,
|
||||
# and dies with "Either datasets or tasks must be provided." — naming neither the path nor
|
||||
# the missing file. Name it here instead. Harbor's own interpreter (the uv-tool venv, per
|
||||
# the shim's shebang) is the only one that can import harbor; skip the check when it or the
|
||||
# helper is absent, so a stripped environment never blocks a runnable task.
|
||||
VALIDATE_TASK_DIR="$SCRIPT_DIR/validate_task_dir.py"
|
||||
# Harbor's own interpreter is the only one that can import harbor. RACCOON_HARBOR_PYTHON
|
||||
# overrides it for tests, which have no harbor to read a shebang from.
|
||||
HARBOR_PY="${RACCOON_HARBOR_PYTHON:-}"
|
||||
if [ -z "$HARBOR_PY" ] && HARBOR_BIN="$(command -v harbor 2>/dev/null)"; then
|
||||
HARBOR_PY="$(sed -n '1s|^#!||p' "$HARBOR_BIN" 2>/dev/null || true)"
|
||||
fi
|
||||
if [ -f "$VALIDATE_TASK_DIR" ] && [ -n "$HARBOR_PY" ] && [ -x "${HARBOR_PY%% *}" ]; then
|
||||
# --install-only implies --disable-verification in harbor, for task validation too.
|
||||
VALIDATE_ARGS=()
|
||||
case " $* " in
|
||||
*" --disable-verification "* | *" --install-only "*) VALIDATE_ARGS+=(--disable-verification) ;;
|
||||
esac
|
||||
VALIDATE_ERR="$(mktemp)"
|
||||
# Gate on the printed verdict, never the exit code: an interpreter that cannot run
|
||||
# the helper at all (a stub harbor with a bash shebang) also exits non-zero, and a
|
||||
# preflight that can refuse a run must never refuse one that would have worked.
|
||||
VALIDATE_OUT="$("$HARBOR_PY" "$VALIDATE_TASK_DIR" "$TASK_DIR" \
|
||||
${VALIDATE_ARGS[@]+"${VALIDATE_ARGS[@]}"} 2>"$VALIDATE_ERR")" || true
|
||||
if [ "$VALIDATE_OUT" = "verdict=invalid" ]; then
|
||||
echo "Error: harbor will not accept $TASK_DIR as a task." >&2
|
||||
sed 's|^| |' "$VALIDATE_ERR" >&2
|
||||
echo "Fix the file named above, then re-run; harbor's own error names no file." >&2
|
||||
if [ -d "$REPO_ROOT/harbor-tasks/_task-scaffold" ]; then
|
||||
echo "A missing toolkit-managed file (tests/test.sh, tests/render-grade-consolidated.py)" >&2
|
||||
echo "can be copied from harbor-tasks/_task-scaffold/ at the same relative path. Never" >&2
|
||||
echo "overwrite a task.toml or instruction.md you have already written — repair it." >&2
|
||||
fi
|
||||
rm -f "$VALIDATE_ERR"
|
||||
exit 1
|
||||
fi
|
||||
rm -f "$VALIDATE_ERR"
|
||||
fi
|
||||
|
||||
# Preflight: recompute the browser marker from task.toml.
|
||||
#
|
||||
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
|
||||
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
|
||||
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
|
||||
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
|
||||
# by leaving it alone.
|
||||
BROWSER_OPTIN=0
|
||||
if [ -f "$TASK_DIR/task.toml" ] &&
|
||||
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
|
||||
BROWSER_OPTIN=1
|
||||
fi
|
||||
if [ -d "$TASK_DIR/environment" ]; then
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
|
||||
fi
|
||||
|
||||
# Preflight: restage the DNS jail script, for the same reason as the marker above.
|
||||
#
|
||||
# The Dockerfile COPYs environment/dns-jail/, and a missing COPY source fails the BUILD --
|
||||
# which would kill every trial on the task, the one outcome the jail must never cause. A
|
||||
# context staged before this existed passes the workspace guard above and would then die at
|
||||
# build, so recreate the directory here and refill it when the source is around. Derived,
|
||||
# so there is nothing to preserve by leaving it alone.
|
||||
if [ -d "$TASK_DIR/environment" ]; then
|
||||
mkdir -p "$TASK_DIR/environment/dns-jail"
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
|
||||
break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Preflight: report — never block — on edits to toolkit-managed files.
|
||||
#
|
||||
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt-consolidated.md ship from
|
||||
# task-shared/ and decide how the trial runs and how the grade is produced, so an edit
|
||||
# makes a task's runs hard to compare with the rest. Surface that here, before a trial
|
||||
# burns agent time. It is advisory on purpose: an author who edited one did it to get
|
||||
# unstuck, and refusing to run their trial punishes a misunderstanding. `|| true` also
|
||||
# means a checker that can't run (a fresh unzip with no node_modules) never reads as an
|
||||
# edit. The checker only exists in the worker toolkit; here the file is absent.
|
||||
CHECK_INFRA="$REPO_ROOT/scripts/check-task-infra.ts"
|
||||
if [ -f "$CHECK_INFRA" ]; then
|
||||
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
|
||||
fi
|
||||
|
||||
# Preflight: warn — loudly, but never block — when the live workspace has
|
||||
# changes that a rebuild from the pinned commit + workspace.patch would lose.
|
||||
# Trials run against the live workspace, but the finalized task keeps only the
|
||||
# rebuild inputs (the workspace/ dir is gitignored) and every downstream
|
||||
# consumer rebuilds from them, so anything uncaptured silently vanishes after
|
||||
# packaging.
|
||||
# The helper sits next to this script in a packed toolkit and under the
|
||||
# toolkit's static scripts in the internal repo layout.
|
||||
for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \
|
||||
"$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do
|
||||
if [ -f "$WS_SYNC" ]; then
|
||||
bash "$WS_SYNC" "$TASK_DIR" || true
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
# Agent + model selection, from scripts/harness-registry.toml via resolve_harness.
|
||||
# The agent classes come from scripts/{snapshot,codex,gemini}_agent.py or
|
||||
# harness_agents.py (hence the PYTHONPATH). The Claude variants reuse the claude
|
||||
# binary baked into the task image instead of re-downloading it at agent-setup —
|
||||
# stock claude-code's runtime download (~240 MB) races the 360s agent-setup timeout
|
||||
# and loses on slow-egress hosts (AgentSetupTimeoutError). Tasks that ship a
|
||||
# non-empty environment/session.jsonl additionally resume the staged session;
|
||||
# single-turn tasks get the non-resuming class.
|
||||
#
|
||||
# resolve_harness exits non-zero (and prints why) when the selection could not
|
||||
# produce a usable grade — an unknown/disabled harness, one that writes no ATIF
|
||||
# trajectory, a missing credential, or a task needing resume on a harness that
|
||||
# can't. Failing here costs a second; failing later costs the whole trial, and the
|
||||
# resume case wouldn't fail at all, it would silently grade the wrong thing.
|
||||
# _raccoon_python comes from lib/harness-credentials.sh, sourced above. Define a fallback
|
||||
# only if that file was missing, so the error below is about the interpreter rather than an
|
||||
# unbound function.
|
||||
command -v _raccoon_python >/dev/null 2>&1 || _raccoon_python() { return 1; }
|
||||
|
||||
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
|
||||
RACCOON_PY=$(_raccoon_python) || {
|
||||
echo "harbor-run: ERROR — no python3.11+ with tomllib on PATH, so the harness registry" >&2
|
||||
echo "harbor-run: cannot be read and the agent class cannot be resolved. Set" >&2
|
||||
echo "harbor-run: RACCOON_PYTHON to an interpreter that has tomllib (3.11+)." >&2
|
||||
exit 1
|
||||
}
|
||||
RESOLVED="$("$RACCOON_PY" "$SCRIPT_DIR/resolve_harness.py" \
|
||||
--task-dir "$TASK_DIR" \
|
||||
${HARNESS_ARGS[@]+"${HARNESS_ARGS[@]}"})" || exit 1
|
||||
eval "$RESOLVED"
|
||||
# --agent-import-path, not --agent: harbor 0.20 deprecates it but still accepts it, and it
|
||||
# is the flag whose recorded shape (`agents[0].import_path`) every run on disk and every
|
||||
# reader expects. --agent leaves import_path null and puts the class in `name`, which
|
||||
# silently empties the agent field in published benchmark rows. Revisit if the pin moves.
|
||||
AGENT_FLAGS="--agent-import-path $AGENT_IMPORT_PATH"
|
||||
# Empty EFFORT_KWARG means "run the harness's native default config" — pass no
|
||||
# effort kwarg at all rather than an empty one, which harbor would reject.
|
||||
EFFORT_FLAGS=""
|
||||
[ -n "$EFFORT_KWARG" ] && EFFORT_FLAGS="--ak $EFFORT_KWARG=$EFFORT_VALUE"
|
||||
# FAST_KWARG is non-empty only when --fast was passed AND the harness declares one
|
||||
# (resolve_harness refuses the flag otherwise).
|
||||
FAST_FLAGS=""
|
||||
[ -n "${FAST_KWARG:-}" ] && FAST_FLAGS="--ak $FAST_KWARG=true"
|
||||
# The kwarg above reaches the trial agent only; the verifier launches its own grader
|
||||
# claude, which the task's tests/test.sh puts in fast mode from GRADER_FAST_MODE.
|
||||
GRADER_FAST_FLAGS=""
|
||||
[ -n "$FAST_REQUESTED" ] && GRADER_FAST_FLAGS="--verifier-env GRADER_FAST_MODE=true"
|
||||
|
||||
# Environment backend. An explicit HARBOR_ENV always wins (either direction).
|
||||
# Otherwise the default is context-dependent:
|
||||
# - daytona for the internal repo: runs the trial in a cloud sandbox over
|
||||
# HTTP, so it needs no local docker daemon and works *inside* the primary
|
||||
# devcontainer. (HARBOR_ENV=docker uses the host docker daemon instead —
|
||||
# free and offline, but host-only; the devcontainer has no docker.sock.)
|
||||
# - docker for the worker toolkit: it's provisioned only for the local docker
|
||||
# backend (docker CLI + bind-mounted docker.sock, no DAYTONA_API_KEY), so a
|
||||
# daytona default would just error out. We detect a toolkit checkout by
|
||||
# toolkit.json at the repo root — a file the packaging step writes that the
|
||||
# internal repo never has. It lives in the bind-mounted workspace, not a
|
||||
# baked image layer, so this holds even when the container's HARBOR_ENV pin
|
||||
# is missing (e.g. a stale, pre-pin image).
|
||||
if [ -n "${HARBOR_ENV:-}" ]; then
|
||||
ENV_TYPE="$HARBOR_ENV"
|
||||
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
|
||||
ENV_TYPE="docker"
|
||||
else
|
||||
ENV_TYPE="daytona"
|
||||
fi
|
||||
|
||||
# --no-delete keeps the environment around after the trial for inspection.
|
||||
# That's free for a local docker container, but a Daytona or Modal sandbox is
|
||||
# *billed* while it exists — keeping it would leak a paid sandbox on every run.
|
||||
# Harbor downloads the trial logs into harbor-jobs before teardown either way, so
|
||||
# for the cloud backends we let it delete the sandbox; for docker we keep the container.
|
||||
DELETE_FLAGS="--no-delete"
|
||||
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
|
||||
|
||||
# Orphan resilience (daytona): harbor tears sandboxes down per-trial + via an
|
||||
# atexit that only closes the client — neither runs on SIGTERM/SIGKILL/crash, so
|
||||
# a killed run leaks STARTED sandboxes that hog the shared pool until (if ever)
|
||||
# an account default reaps them. Tell Daytona to auto-stop an IDLE sandbox after
|
||||
# 20 min (auto-delete on stop), so orphans self-clean however the process dies.
|
||||
# Safe for live trials: a running agent/grader keeps the sandbox active.
|
||||
AUTOSTOP_FLAGS=""
|
||||
[ "$ENV_TYPE" = "daytona" ] && AUTOSTOP_FLAGS="--ek auto_stop_interval_mins=20"
|
||||
|
||||
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
|
||||
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
|
||||
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
|
||||
RESOURCE_FLAGS=""
|
||||
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
|
||||
|
||||
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
|
||||
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
|
||||
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Task-image Claude Code floor. A task image installs Claude Code when it is first built and
|
||||
# Docker reuses that layer on every later build, --force-build included, so an image built
|
||||
# before the grader model's minimum CLI shipped fails every grade with "does not support this
|
||||
# model" and the trial ends in RewardFileNotFoundError. Before a local docker trial, check the
|
||||
# hb__ task images on this daemon and remove any that are too old, together with the build
|
||||
# cache, so harbor's build below installs a current CLI. Advisory: no docker, no daemon, no
|
||||
# images, or RACCOON_SKIP_IMAGE_PREFLIGHT=1 means nothing happens. The floor follows
|
||||
# GRADER_CLI_MIN in the shared test.sh; RACCOON_CLAUDE_CODE_MIN overrides it.
|
||||
CLAUDE_CODE_MIN="${RACCOON_CLAUDE_CODE_MIN:-}"
|
||||
if [ -z "$CLAUDE_CODE_MIN" ]; then
|
||||
for _ts in "$REPO_ROOT/task-shared/test.sh" "$REPO_ROOT/harbor-tasks/raccoon-shared/test.sh"; do
|
||||
[ -f "$_ts" ] || continue
|
||||
CLAUDE_CODE_MIN="$(sed -n 's/^GRADER_CLI_MIN=.*:-\([0-9][0-9.]*\)}.*/\1/p' "$_ts" | head -1)"
|
||||
[ -n "$CLAUDE_CODE_MIN" ] && break
|
||||
done
|
||||
fi
|
||||
CLAUDE_CODE_MIN="${CLAUDE_CODE_MIN:-2.1.251}"
|
||||
if [ "$ENV_TYPE" = "docker" ] && [ "${RACCOON_SKIP_IMAGE_PREFLIGHT:-0}" != "1" ] \
|
||||
&& command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
|
||||
STALE_IMAGES=""
|
||||
for img in $(docker images --format '{{.Repository}}:{{.Tag}}' 2>/dev/null | grep '^hb__' || true); do
|
||||
v="$(docker run --rm --entrypoint claude "$img" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
|
||||
[ -n "$v" ] || continue
|
||||
if [ "$(printf '%s\n%s\n' "$CLAUDE_CODE_MIN" "$v" | sort -V | head -1)" != "$CLAUDE_CODE_MIN" ]; then
|
||||
STALE_IMAGES="$STALE_IMAGES $img"
|
||||
echo "Task image $img carries Claude Code $v; the grader needs $CLAUDE_CODE_MIN or newer. Removing it so the next build installs a current one." >&2
|
||||
fi
|
||||
done
|
||||
if [ -n "$STALE_IMAGES" ]; then
|
||||
for img in $STALE_IMAGES; do
|
||||
# Trial containers harbor kept for inspection pin the image; drop them first.
|
||||
for c in $(docker ps -aq --filter "ancestor=$img" 2>/dev/null); do docker rm -f "$c" >/dev/null 2>&1 || true; done
|
||||
docker image rm -f "$img" >/dev/null 2>&1 || true
|
||||
done
|
||||
# The stale install layer also lives in the build cache, where a rebuild would find it.
|
||||
docker builder prune -af >/dev/null 2>&1 || true
|
||||
echo "The task image rebuilds once with a current Claude Code; later runs reuse it." >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# Launch-time input-checksum capture. Snapshot the task inputs BEFORE harbor
|
||||
# starts, so the recorded hashes are what the agent actually ran against —
|
||||
# an input edited between this run and copy-reference-run no longer records
|
||||
# post-edit state and masks staleness. The capture is stamped into the trial
|
||||
# dirs after the run (below); copy-reference-run prefers it over its weaker
|
||||
# capture-at-copy fallback. Advisory end to end: a failure here never blocks
|
||||
# the run.
|
||||
#
|
||||
# The post-run stamping watches the default `harbor-jobs` output dir. If the
|
||||
# caller overrides the output dir via extra args, we can't know where the
|
||||
# trials will land — skip stamping and say so, rather than silently stamping
|
||||
# nothing (runs then fall back to copy-time capture in copy-reference-run).
|
||||
STAMP_FILE=""
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
-o|--output*)
|
||||
echo "Note: custom harbor output dir passed ($arg) — skipping launch-time input-checksum stamping; reference runs will fall back to copy-time capture." >&2
|
||||
STAMP_FILE="skip"
|
||||
break
|
||||
;;
|
||||
esac
|
||||
done
|
||||
if [ "$STAMP_FILE" != "skip" ]; then
|
||||
STAMP_FILE="$(mktemp)"
|
||||
if ! npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" capture "$TASK_DIR" --out "$STAMP_FILE"; then
|
||||
echo "Warning: could not capture launch-time input checksums (staleness will be judged from copy-time capture instead)" >&2
|
||||
rm -f "$STAMP_FILE"
|
||||
STAMP_FILE=""
|
||||
fi
|
||||
else
|
||||
STAMP_FILE=""
|
||||
fi
|
||||
JOBS_BEFORE="$(ls -1 harbor-jobs 2>/dev/null || true)"
|
||||
|
||||
# DNS jail: the trial resolves the LLM proxy and nothing else. Opt-in with
|
||||
# RACCOON_DNS_JAIL=1, which .env can set; the agent applies it inside the container from
|
||||
# the proxy URL it already holds, so nothing is passed on the harbor command line.
|
||||
#
|
||||
# Claude and codex only (see scripts/dnsjail.py): both are baked into the task images,
|
||||
# while every other harness downloads its CLI at agent-setup. Say so out loud — a silently
|
||||
# unjailed trial is worse than no jail, because the caller believes otherwise.
|
||||
if [ "${RACCOON_DNS_JAIL:-0}" = "1" ]; then
|
||||
case "$AGENT_IMPORT_PATH" in
|
||||
snapshot_agent:* | codex_agent:*)
|
||||
# The script is baked into the image, so a task frozen before this feature has
|
||||
# nothing to invoke. Most hand-authored per-task Dockerfiles are in that group.
|
||||
if [ -f "$TASK_DIR/environment/Dockerfile" ] &&
|
||||
! grep -q 'raccoon-dns-jail' "$TASK_DIR/environment/Dockerfile"; then
|
||||
echo "Note: DNS jail skipped — this task's image predates it and ships no" \
|
||||
"resolver. This trial has full network access." >&2
|
||||
fi
|
||||
;;
|
||||
*) echo "Note: DNS jail skipped — it is applied by the claude and codex agents" \
|
||||
"only. This trial has full network access." >&2 ;;
|
||||
esac
|
||||
fi
|
||||
|
||||
# The model comes from the registry (already resolved above) and is always a
|
||||
# CONCRETE id, never a shorthand alias. On the manual (non-snapshot) path the agent
|
||||
# passes the model via ANTHROPIC_MODEL, where a shorthand is NOT alias-resolved, so
|
||||
# the configured base-URL endpoint rejects it (400 "Invalid model: <shorthand>") and
|
||||
# trials die on turn 1. Bump `default_model` in scripts/harness-registry.toml when a
|
||||
# newer model ships.
|
||||
#
|
||||
# harbor runs as a child (this script used to `exec` it, but the post-run
|
||||
# stamping needs to run after harbor exits), so forward TERM/INT: a `kill`
|
||||
# aimed at this wrapper's PID must take harbor down with it, not orphan a
|
||||
# running job (this repo has been bitten by zombie harbor coordinators
|
||||
# before).
|
||||
HARBOR_EXIT=0
|
||||
HARBOR_SIGNALLED=""
|
||||
harbor run \
|
||||
-p "$TASK_DIR" \
|
||||
$AGENT_FLAGS \
|
||||
-m "$MODEL" \
|
||||
-e "$ENV_TYPE" \
|
||||
$DELETE_FLAGS \
|
||||
$AUTOSTOP_FLAGS \
|
||||
$RESOURCE_FLAGS \
|
||||
--yes \
|
||||
-o harbor-jobs \
|
||||
$EFFORT_FLAGS \
|
||||
$FAST_FLAGS \
|
||||
$GRADER_FAST_FLAGS \
|
||||
"$@" &
|
||||
HARBOR_PID=$!
|
||||
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
|
||||
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
if [ -n "$HARBOR_SIGNALLED" ]; then
|
||||
# The first wait was interrupted by the trap; wait again so harbor's real
|
||||
# exit status (not the shell's signal status) is what we propagate.
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
fi
|
||||
trap - TERM INT
|
||||
|
||||
# Stamp the launch-time capture into this task's trial dirs in the job dir(s)
|
||||
# this run created (the harbor-jobs entries that didn't exist before the
|
||||
# run). Harbor names job dirs with a timestamp, so new entries are this run's
|
||||
# output — plus, when several harbor-runs share a cwd, possibly a concurrent
|
||||
# run's; `apply` is slug-scoped so another task's trials are never stamped
|
||||
# with this task's inputs.
|
||||
if [ -n "$STAMP_FILE" ]; then
|
||||
NEW_JOBS="$(comm -13 <(printf '%s\n' "$JOBS_BEFORE" | sort) <(ls -1 harbor-jobs 2>/dev/null | sort) | sed 's|^|harbor-jobs/|')"
|
||||
if [ -n "$NEW_JOBS" ]; then
|
||||
# shellcheck disable=SC2086 # job-dir names are timestamps, never spaced
|
||||
npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" apply "$STAMP_FILE" $NEW_JOBS || \
|
||||
echo "Warning: could not stamp trial dirs with launch-time input checksums" >&2
|
||||
fi
|
||||
rm -f "$STAMP_FILE"
|
||||
fi
|
||||
|
||||
# Repeat the toolkit-managed-file notice AFTER the trial. The preflight copy is
|
||||
# minutes of harbor output up the scrollback by now, which for a notice nothing
|
||||
# enforces means nobody reads it. This one lands where the author is looking.
|
||||
if [ -f "$CHECK_INFRA" ]; then
|
||||
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
|
||||
fi
|
||||
|
||||
exit "$HARBOR_EXIT"
|
||||
@@ -0,0 +1,271 @@
|
||||
version = 1
|
||||
|
||||
[[harness]]
|
||||
id = "claude-code"
|
||||
label = "Claude Code"
|
||||
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
|
||||
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
|
||||
# given a browser can look at the screenshot it just took. Distinct classes with distinct
|
||||
# names, because a different toolset is a different agent.
|
||||
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
|
||||
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
|
||||
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
|
||||
import_path_aliases = [
|
||||
"snapshot_agent:FullToolsetSnapshotClaudeCode",
|
||||
"snapshot_agent:FullToolsetPreinstalledClaudeCode",
|
||||
"harbor.agents.installed.claude_code:ClaudeCode",
|
||||
]
|
||||
legacy_bare_model_rows = true
|
||||
default_model = "claude-opus-5[1m]"
|
||||
model_id_shape = "bare"
|
||||
effort_kwarg = "reasoning_effort"
|
||||
effort_default = "max"
|
||||
fast_kwarg = "fast_mode"
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = true
|
||||
seed_native = true
|
||||
seed_atif = true
|
||||
authoring = true
|
||||
cli = "claude"
|
||||
install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash && break; echo \"claude install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
|
||||
# No agent_config: claude reduces its toolset with `--tools`, not `-c key=value`, so the
|
||||
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
|
||||
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
|
||||
# unlike model and effort, which are interpolated from this row.
|
||||
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
|
||||
|
||||
[[harness]]
|
||||
id = "codex"
|
||||
label = "OpenAI Codex CLI"
|
||||
agent_import_path = "codex_agent:NativeSnapshotCodex"
|
||||
agent_import_path_single_turn = "codex_agent:SystemNodeCodex"
|
||||
import_path_aliases = [
|
||||
"codex_agent:InlineSnapshotCodex",
|
||||
"harbor.agents.installed.codex:Codex",
|
||||
]
|
||||
legacy_bare_model_rows = true
|
||||
default_model = "gpt-5.6-sol"
|
||||
model_id_shape = "bare"
|
||||
effort_kwarg = "reasoning_effort"
|
||||
effort_default = "max"
|
||||
key_env = "OPENAI_API_KEY"
|
||||
base_url_env = "OPENAI_BASE_URL"
|
||||
proxy_path = "openai/v1"
|
||||
writes_atif = true
|
||||
capture = true
|
||||
seed_native = true
|
||||
seed_atif = true
|
||||
authoring = true
|
||||
cli = "codex"
|
||||
install = "for i in 1 2 3; do curl -fsSL https://chatgpt.com/codex/install.sh | CODEX_NON_INTERACTIVE=1 sh && break; echo \"codex install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
|
||||
skills_dir = "$HOME/.agents/skills"
|
||||
config_path = "${CODEX_HOME:-$HOME/.codex}/config.toml"
|
||||
auth_path = "${CODEX_HOME:-$HOME/.codex}/auth.json"
|
||||
auth_key_env = "OPENAI_API_KEY"
|
||||
agent_config = """
|
||||
web_search = "disabled"
|
||||
|
||||
[agents]
|
||||
enabled = false
|
||||
|
||||
[tools]
|
||||
update_plan = { enabled = false }
|
||||
experimental_request_user_input = { enabled = false }
|
||||
|
||||
[features]
|
||||
goals = false
|
||||
multi_agent = false
|
||||
multi_agent_v2 = false
|
||||
memories = false
|
||||
external_agent_memory_import = false
|
||||
"""
|
||||
container_config = """
|
||||
openai_base_url = "${OPENAI_BASE_URL}"
|
||||
"""
|
||||
explore_config = """
|
||||
[hooks]
|
||||
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
|
||||
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
|
||||
"""
|
||||
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
|
||||
|
||||
[[harness]]
|
||||
id = "gemini-cli"
|
||||
label = "Gemini CLI"
|
||||
agent_import_path = "gemini_agent:NativeSnapshotGeminiCli"
|
||||
agent_import_path_single_turn = "gemini_agent:SystemNodeGeminiCli"
|
||||
import_path_aliases = ["harbor.agents.installed.gemini_cli:GeminiCli"]
|
||||
legacy_bare_model_rows = true
|
||||
default_model = "gemini-3.5-flash"
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = "reasoning_effort"
|
||||
effort_default = "high"
|
||||
key_env = "GEMINI_API_KEY"
|
||||
base_url_env = "GEMINI_API_BASE"
|
||||
proxy_path = "gemini"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = true
|
||||
seed_atif = false
|
||||
|
||||
[[harness]]
|
||||
id = "antigravity-cli"
|
||||
label = "Antigravity CLI"
|
||||
agent_import_path = "harness_agents:BenchAntigravity"
|
||||
import_path_aliases = ["harbor.agents.installed.antigravity_cli:AntigravityCli"]
|
||||
legacy_bare_model_rows = false
|
||||
# The prefix is load-bearing: harbor's adapter raises without a "/" in the id.
|
||||
# agy carries its own model catalogue and DROPS entries between point releases
|
||||
# (1.1.25 removed gemini-3.5-flash, breaking every run). If trials start failing
|
||||
# with "not recognized as a known model", run `agy --model bogus --prompt=x` to
|
||||
# print the current catalogue and update this.
|
||||
default_model = "google/gemini-3.8-flash"
|
||||
model_id_shape = "provider/model"
|
||||
# Not optional: agy refuses a Gemini 3 model with no --effort ("requires --effort
|
||||
# (available: low, medium, high)"). low/high are safe on pro and flash alike.
|
||||
effort_kwarg = "reasoning_effort"
|
||||
effort_default = "high"
|
||||
key_env = "GEMINI_API_KEY"
|
||||
base_url_env = "GOOGLE_GEMINI_BASE_URL"
|
||||
proxy_path = "gemini"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
# agy cannot be handed externally-produced history, so multi-turn tasks must
|
||||
# hard-fail rather than silently run cold. See work-logs/antigravity-harness.md.
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
|
||||
[[harness]]
|
||||
id = "opencode"
|
||||
label = "OpenCode"
|
||||
agent_import_path = "harness_agents:BenchOpenCode"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = ""
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
flaky_hangs = true
|
||||
|
||||
[[harness]]
|
||||
id = "goose"
|
||||
label = "Goose"
|
||||
agent_import_path = "harness_agents:BenchGoose"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = ""
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
|
||||
[[harness]]
|
||||
id = "mini-swe-agent"
|
||||
label = "mini-swe-agent"
|
||||
agent_import_path = "harness_agents:BenchMiniSweAgent"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = ""
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
|
||||
[[harness]]
|
||||
id = "cline-cli"
|
||||
label = "Cline CLI"
|
||||
agent_import_path = "harness_agents:BenchCline"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider:model"
|
||||
effort_kwarg = ""
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
|
||||
[[harness]]
|
||||
id = "crush"
|
||||
label = "Crush"
|
||||
agent_import_path = "harness_agents:Crush"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = ""
|
||||
key_env = "ANTHROPIC_API_KEY"
|
||||
base_url_env = "ANTHROPIC_BASE_URL"
|
||||
proxy_path = "anthropic"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
flaky_hangs = true
|
||||
|
||||
[[harness]]
|
||||
id = "amp"
|
||||
label = "Amp"
|
||||
agent_import_path = "harness_agents:Amp"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "bare"
|
||||
effort_kwarg = ""
|
||||
key_env = "AMP_API_KEY"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
enabled = false
|
||||
|
||||
[[harness]]
|
||||
id = "cursor-cli"
|
||||
label = "Cursor CLI"
|
||||
agent_import_path = "harness_agents:BenchCursorCli"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "bare"
|
||||
effort_kwarg = ""
|
||||
key_env = "CURSOR_API_KEY"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
enabled = false
|
||||
|
||||
[[harness]]
|
||||
id = "copilot-cli"
|
||||
label = "GitHub Copilot CLI"
|
||||
agent_import_path = "harness_agents:BenchCopilotCli"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "bare"
|
||||
effort_kwarg = ""
|
||||
key_env = "GITHUB_TOKEN"
|
||||
writes_atif = true
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
enabled = false
|
||||
|
||||
[[harness]]
|
||||
id = "aider"
|
||||
label = "Aider"
|
||||
agent_import_path = "harness_agents:BenchAider"
|
||||
legacy_bare_model_rows = false
|
||||
model_id_shape = "provider/model"
|
||||
effort_kwarg = ""
|
||||
writes_atif = false
|
||||
capture = false
|
||||
seed_native = false
|
||||
seed_atif = false
|
||||
enabled = false
|
||||
@@ -0,0 +1,29 @@
|
||||
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
|
||||
// real shape instead of `any`.
|
||||
|
||||
export interface Turn {
|
||||
/** Line index in the native session file. */
|
||||
index: number;
|
||||
role: 'user' | 'assistant';
|
||||
text: string;
|
||||
/** A slash-command turn, not real conversation. */
|
||||
isCommand: boolean;
|
||||
/** This record concluded its turn — the truncation boundary. */
|
||||
endsTurn: boolean;
|
||||
}
|
||||
|
||||
export interface Session {
|
||||
harness: string;
|
||||
rawPath: string;
|
||||
/** The harness own id for this conversation. */
|
||||
sessionId: string | null;
|
||||
lines: string[];
|
||||
turns: Turn[];
|
||||
}
|
||||
|
||||
export function supportedHarnesses(): string[];
|
||||
export function readSession(harness: string, recordedPath?: string): Session | null;
|
||||
export function truncationIndex(turns: Turn[]): number;
|
||||
export function turnsFromLines(harness: string, lines: string[]): Turn[];
|
||||
export function linearSnapshotLines(session: Session, startLine?: number): string[];
|
||||
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];
|
||||
325
worker-toolkit-potion-polyglot-orig/scripts/harness-session.mjs
Normal file
325
worker-toolkit-potion-polyglot-orig/scripts/harness-session.mjs
Normal file
@@ -0,0 +1,325 @@
|
||||
// Locate and read a harness's native conversation, so capture-snapshot can work
|
||||
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
|
||||
// annotation, metadata) is harness-agnostic.
|
||||
//
|
||||
// The returned session stays in the harness's OWN native format: the seeding design
|
||||
// hands a native blob back to the same harness, and codex_agent reads the same staged
|
||||
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
|
||||
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
|
||||
/**
|
||||
* @typedef {object} Turn
|
||||
* @property {number} index line index in the native session file
|
||||
* @property {'user'|'assistant'} role
|
||||
* @property {string} text
|
||||
* @property {boolean} isCommand a slash-command turn, not real conversation
|
||||
* @property {boolean} endsTurn this record concluded its turn
|
||||
*/
|
||||
|
||||
/**
|
||||
* @typedef {object} Session
|
||||
* @property {string} harness
|
||||
* @property {string} rawPath
|
||||
* @property {string|null} sessionId the harness's own id for this conversation
|
||||
* @property {string[]} lines
|
||||
* @property {Turn[]} turns
|
||||
*/
|
||||
|
||||
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
|
||||
// answer is stable — two sessions written in the same millisecond are common.
|
||||
function newestUnder(root, matches) {
|
||||
if (!fs.existsSync(root)) return null;
|
||||
const found = [];
|
||||
const walk = (dir) => {
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
for (const entry of entries) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) walk(full);
|
||||
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
|
||||
}
|
||||
};
|
||||
walk(root);
|
||||
if (found.length === 0) return null;
|
||||
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
|
||||
return found[0].full;
|
||||
}
|
||||
|
||||
// User-role records codex writes that the human did not type: its own environment
|
||||
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
|
||||
// Matched only at the START of the text, so a turn that merely quotes one is still real
|
||||
// conversation.
|
||||
function isCodexCommandText(text) {
|
||||
const trimmed = (text || '').trimStart();
|
||||
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
|
||||
return /^\$[\w:.-]+\s*$/.test(trimmed);
|
||||
}
|
||||
|
||||
const HARNESSES = {
|
||||
'claude-code': {
|
||||
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
|
||||
findSession() {
|
||||
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
|
||||
n.endsWith('.jsonl')
|
||||
);
|
||||
},
|
||||
|
||||
/** Claude names the transcript for its session id. */
|
||||
sessionId(rawPath) {
|
||||
return path.basename(rawPath, '.jsonl');
|
||||
},
|
||||
/**
|
||||
* One turn per conversational record. `endsTurn` marks an assistant record that
|
||||
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
|
||||
* file-history, permission-mode) carry no role and are skipped.
|
||||
*/
|
||||
/** @param {string[]} lines @returns {Turn[]} */
|
||||
readTurns(lines) {
|
||||
/** @type {Turn[]} */
|
||||
const turns = [];
|
||||
for (const [index, line] of lines.entries()) {
|
||||
let entry;
|
||||
try {
|
||||
entry = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const role =
|
||||
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
|
||||
if (!role) continue;
|
||||
const content = entry.message?.content;
|
||||
const text =
|
||||
typeof content === 'string'
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? content
|
||||
.filter((b) => b && b.type === 'text')
|
||||
.map((b) => b.text ?? '')
|
||||
.join('')
|
||||
: '';
|
||||
turns.push({
|
||||
index,
|
||||
role,
|
||||
text,
|
||||
isCommand:
|
||||
role === 'user' &&
|
||||
typeof content === 'string' &&
|
||||
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
|
||||
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
|
||||
});
|
||||
}
|
||||
return turns;
|
||||
},
|
||||
},
|
||||
|
||||
codex: {
|
||||
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
|
||||
findSession() {
|
||||
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
|
||||
return newestUnder(
|
||||
path.join(home, 'sessions'),
|
||||
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
|
||||
);
|
||||
},
|
||||
|
||||
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
|
||||
sessionId(rawPath, lines) {
|
||||
for (const line of lines) {
|
||||
try {
|
||||
const rec = JSON.parse(line);
|
||||
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
},
|
||||
/**
|
||||
* codex rollouts carry `response_item` records whose payload is a message with a
|
||||
* role. An assistant message with no following tool activity ends the turn; codex
|
||||
* records no stop_reason, so a turn ends where the next user message begins —
|
||||
* resolved after the fact below.
|
||||
*/
|
||||
/** @param {string[]} lines @returns {Turn[]} */
|
||||
readTurns(lines) {
|
||||
/** @type {Turn[]} */
|
||||
const turns = [];
|
||||
for (const [index, line] of lines.entries()) {
|
||||
let record;
|
||||
try {
|
||||
record = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (record.type !== 'response_item') continue;
|
||||
const payload = record.payload ?? {};
|
||||
if (payload.type !== 'message') continue;
|
||||
const role =
|
||||
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
|
||||
if (!role) continue;
|
||||
const text = Array.isArray(payload.content)
|
||||
? payload.content.map((b) => b?.text ?? '').join('')
|
||||
: typeof payload.content === 'string'
|
||||
? payload.content
|
||||
: '';
|
||||
turns.push({
|
||||
index,
|
||||
role,
|
||||
text,
|
||||
isCommand: role === 'user' && isCodexCommandText(text),
|
||||
endsTurn: false,
|
||||
});
|
||||
}
|
||||
// An assistant turn ends where the next user turn starts, or at the end.
|
||||
for (let i = 0; i < turns.length; i += 1) {
|
||||
if (turns[i].role !== 'assistant') continue;
|
||||
const next = turns[i + 1];
|
||||
turns[i].endsTurn = !next || next.role === 'user';
|
||||
}
|
||||
return turns;
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
/** @returns {string[]} */
|
||||
export function supportedHarnesses() {
|
||||
return Object.keys(HARNESSES);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the current session for `harness`. Returns null when nothing is found, so the
|
||||
* caller can report which harness had no conversation to capture.
|
||||
*/
|
||||
/**
|
||||
* @param {string} harness
|
||||
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
|
||||
* over the newest-file scan, which can pick a different session in a busy container.
|
||||
* @returns {Session | null}
|
||||
*/
|
||||
export function readSession(harness, recordedPath) {
|
||||
const reader = HARNESSES[harness];
|
||||
if (!reader) {
|
||||
throw new Error(
|
||||
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
|
||||
);
|
||||
}
|
||||
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
|
||||
if (!rawPath) return null;
|
||||
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
|
||||
return {
|
||||
harness,
|
||||
rawPath,
|
||||
lines,
|
||||
turns: reader.readTurns(lines),
|
||||
sessionId: reader.sessionId(rawPath, lines),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Index of the last record to keep: the last turn-ending assistant record before the
|
||||
* final real user turn. Drops the prompt that elicited the failure and the failure
|
||||
* response, so the test agent inherits context but not the answer.
|
||||
*
|
||||
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
|
||||
* treat as "seed nothing and run cold".
|
||||
*/
|
||||
/**
|
||||
* @param {Turn[]} turns
|
||||
* @returns {number}
|
||||
*/
|
||||
export function truncationIndex(turns) {
|
||||
let lastUser = -1;
|
||||
for (const turn of turns) {
|
||||
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
|
||||
}
|
||||
if (lastUser < 0) return -1;
|
||||
let cut = -1;
|
||||
for (const turn of turns) {
|
||||
if (turn.index >= lastUser) break;
|
||||
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
|
||||
}
|
||||
return cut;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse already-read lines with a harness's reader, for callers that have the text
|
||||
* rather than a path.
|
||||
*
|
||||
* @param {string} harness
|
||||
* @param {string[]} lines
|
||||
* @returns {Turn[]}
|
||||
*/
|
||||
export function turnsFromLines(harness, lines) {
|
||||
const reader = HARNESSES[harness];
|
||||
if (!reader) {
|
||||
throw new Error(
|
||||
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
|
||||
);
|
||||
}
|
||||
return reader.readTurns(lines);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
|
||||
* everything up to the snapshot invocation, matching what Claude Code stages when it
|
||||
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
|
||||
* in snapshot-to-task — capture keeps the full conversation.
|
||||
*
|
||||
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
|
||||
* the whole session is kept, which would include the snapshot's own Q&A.
|
||||
*
|
||||
* @param {Session} session
|
||||
* @param {number} [startLine]
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function linearSnapshotLines(session, startLine) {
|
||||
if (typeof startLine === 'number' && startLine >= 0) {
|
||||
return session.lines.slice(0, startLine);
|
||||
}
|
||||
return session.lines;
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop records that describe the AUTHORING container rather than the conversation.
|
||||
*
|
||||
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
|
||||
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
|
||||
* so without this the test agent inherits a list of skills it does not have — one described
|
||||
* as "capture the current conversation and repo state as a snapshot" — and a working
|
||||
* directory that does not exist in the trial. codex re-injects both for the trial, and base
|
||||
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
|
||||
* already re-records with the trial's own cwd; this brings codex to the same place.
|
||||
*
|
||||
* @param {string} harness
|
||||
* @param {string[]} lines
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function stripAuthoringScaffolding(harness, lines) {
|
||||
if (harness === 'claude-code') return lines;
|
||||
return lines.filter((raw) => {
|
||||
let rec;
|
||||
try {
|
||||
rec = JSON.parse(raw);
|
||||
} catch {
|
||||
return true;
|
||||
}
|
||||
const payload = rec?.payload;
|
||||
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
|
||||
const text = (payload.content ?? [])
|
||||
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
|
||||
.join('')
|
||||
.trim();
|
||||
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
|
||||
// worker who quotes one of these strings mid-conversation keeps their turn.
|
||||
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
|
||||
if (payload.role === 'user') return !text.startsWith('<environment_context>');
|
||||
return true;
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
import { existsSync } from 'node:fs';
|
||||
|
||||
(function checkDevcontainer() {
|
||||
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
|
||||
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
|
||||
|
||||
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
|
||||
if (inContainer) return;
|
||||
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
|
||||
if (process.env.CI === 'true' || process.env.CI === '1') return;
|
||||
|
||||
const yellow = '\x1b[33m';
|
||||
const reset = '\x1b[0m';
|
||||
process.stderr.write(
|
||||
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
|
||||
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
|
||||
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
|
||||
);
|
||||
})();
|
||||
@@ -0,0 +1,36 @@
|
||||
"""How codex is handed its API key, kept out of codex_agent so it is testable without
|
||||
harbor (whose venv has no pytest, so anything importing it SKIPs in CI).
|
||||
|
||||
codex reads its key from `$CODEX_HOME/auth.json` and its proxy URL from config.toml —
|
||||
`OPENAI_API_KEY` / `OPENAI_BASE_URL` in the environment are both ignored, verified against
|
||||
0.146.0 and 0.152.0 (an env-var-only run sends no `authorization` header at all).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import shlex
|
||||
|
||||
# Characters harbor's own auth.json writer cannot survive: it interpolates the key into a
|
||||
# shell heredoc, so `"` closes the JSON string and `\` starts an escape.
|
||||
_UNESCAPABLE = '"\\\n\r'
|
||||
|
||||
|
||||
AUTH_JSON_ENV_VAR = "RACCOON_CODEX_AUTH_JSON"
|
||||
|
||||
|
||||
def auth_json_setup(key: str, remote_auth_path: str) -> tuple[dict[str, str], str]:
|
||||
"""The one extra env var — returned separately so it reaches ONLY the setup exec — plus
|
||||
shell writing a parseable auth.json. Subshell: the umask must not outlive this write."""
|
||||
env = {AUTH_JSON_ENV_VAR: json.dumps({"OPENAI_API_KEY": key})}
|
||||
command = (
|
||||
f"(umask 077; printf '%s\\n' \"${AUTH_JSON_ENV_VAR}\" "
|
||||
f">{shlex.quote(remote_auth_path)})\n"
|
||||
)
|
||||
return env, command
|
||||
|
||||
|
||||
def unescapable_chars(key: str) -> list[str]:
|
||||
"""Which characters in `key` harbor's stock heredoc writer would corrupt — empty for
|
||||
every ordinary key, so the caller can refuse instead of 401ing three layers down."""
|
||||
return sorted({c for c in _UNESCAPABLE if c in key})
|
||||
43
worker-toolkit-potion-polyglot-orig/scripts/lib/copy-tree.ts
Normal file
43
worker-toolkit-potion-polyglot-orig/scripts/lib/copy-tree.ts
Normal file
@@ -0,0 +1,43 @@
|
||||
/**
|
||||
* Recursive copy for scripts that must not call `cpSync`: it fails EACCES
|
||||
* against a macOS docker bind mount, where the toolkit's job dirs live.
|
||||
*/
|
||||
|
||||
import {
|
||||
chmodSync,
|
||||
copyFileSync,
|
||||
lstatSync,
|
||||
mkdirSync,
|
||||
readdirSync,
|
||||
readlinkSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
symlinkSync,
|
||||
} from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
/** Copy one entry — symlink, directory or file — preserving its mode. */
|
||||
export function copyPath(src: string, dest: string) {
|
||||
const st = lstatSync(src);
|
||||
if (st.isSymbolicLink()) {
|
||||
rmSync(dest, { force: true });
|
||||
symlinkSync(readlinkSync(src), dest);
|
||||
return;
|
||||
}
|
||||
if (st.isDirectory()) {
|
||||
copyTree(src, dest);
|
||||
return;
|
||||
}
|
||||
// Unlink first: copyFileSync onto an existing file keeps that file's mode.
|
||||
rmSync(dest, { force: true });
|
||||
copyFileSync(src, dest);
|
||||
chmodSync(dest, statSync(src).mode & 0o777);
|
||||
}
|
||||
|
||||
/** Copy `src`'s contents into `dest`, creating `dest` if it doesn't exist. */
|
||||
export function copyTree(src: string, dest: string) {
|
||||
mkdirSync(dest, { recursive: true });
|
||||
for (const entry of readdirSync(src, { withFileTypes: true })) {
|
||||
copyPath(join(src, entry.name), join(dest, entry.name));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
#!/bin/sh
|
||||
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
|
||||
# every other name unresolvable. Runs as root, inside the container.
|
||||
#
|
||||
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
|
||||
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
|
||||
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
|
||||
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
|
||||
#
|
||||
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
|
||||
# applied before it is verified, and any doubt leaves the container's DNS untouched.
|
||||
set -u
|
||||
|
||||
STATE=/tmp/.dnsjail
|
||||
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
|
||||
|
||||
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
|
||||
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
|
||||
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
|
||||
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
|
||||
|
||||
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
|
||||
# later run could mistake for its own filter.
|
||||
drop_ours() {
|
||||
if [ -s "$STATE/dnsmasq.pid" ]; then
|
||||
pid=$(cat "$STATE/dnsmasq.pid")
|
||||
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
|
||||
# some service's child. Confirm it is dnsmasq before signalling it.
|
||||
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
|
||||
dnsmasq) kill "$pid" 2>/dev/null || true ;;
|
||||
esac
|
||||
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
|
||||
# end the caller's shell.
|
||||
dnsjail_apply() {
|
||||
required="${DNSJAIL_ALLOW:-}"
|
||||
extra="${DNSJAIL_ALLOW_EXTRA:-}"
|
||||
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
|
||||
# A blank required list means no model endpoint was found: jailing would strand the agent.
|
||||
set -- $required
|
||||
[ $# -gt 0 ] || return 0
|
||||
|
||||
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
|
||||
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
|
||||
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
|
||||
# silently UNjail a working container.
|
||||
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
|
||||
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
|
||||
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
|
||||
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
# The state dir has to work first: it holds what unjail restores, and a failed write here
|
||||
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
|
||||
# running as the container user in Explore, can drop its own lift markers.
|
||||
mkdir -p "$STATE" 2>/dev/null || return 0
|
||||
chmod 1777 "$STATE" 2>/dev/null || true
|
||||
: > "$STATE/.probe" 2>/dev/null || return 0
|
||||
rm -f "$STATE/.probe" 2>/dev/null || true
|
||||
|
||||
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
|
||||
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
|
||||
# every name.
|
||||
src=/etc/resolv.conf
|
||||
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
|
||||
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
|
||||
[ "$up" = "127.0.0.1" ] && up=""
|
||||
|
||||
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
|
||||
srv=""
|
||||
for h in $allow; do srv="$srv --server=/$h/$up"; done
|
||||
drop_ours
|
||||
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
|
||||
# one would rather than an answer this resolver decided to keep.
|
||||
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
|
||||
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
|
||||
>/dev/null 2>>"$STATE/dnsmasq.err" || true
|
||||
fi
|
||||
|
||||
# Ask the resolver directly: the model endpoint must answer and the control must not --
|
||||
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
|
||||
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
|
||||
# through the catch-all, and one of those must not silently disable the whole jail.
|
||||
live=1
|
||||
for h in $required; do
|
||||
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
|
||||
done
|
||||
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
|
||||
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
|
||||
# resolve through the catch-all, and must not take the whole jail down with it.
|
||||
if [ -n "$live" ]; then
|
||||
for h in $extra; do
|
||||
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
|
||||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
|
||||
done
|
||||
fi
|
||||
|
||||
if [ -z "$live" ]; then
|
||||
# Say why. A silent decline is indistinguishable from a jail that worked, and the
|
||||
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
|
||||
# AF_NETLINK, so dnsmasq cannot start there at all).
|
||||
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
|
||||
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
|
||||
drop_ours
|
||||
# Failing open has to mean actually open, including when an earlier run left this
|
||||
# container jailed.
|
||||
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
|
||||
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
|
||||
# would leave unjail a permanent no-op.
|
||||
if ! jailed_now; then
|
||||
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
|
||||
fi
|
||||
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
|
||||
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
|
||||
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
|
||||
rm -rf "$STATE/lifts" 2>/dev/null || true
|
||||
|
||||
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
|
||||
# which means the replacement has to be complete BEFORE the write starts. Keep every
|
||||
# non-nameserver directive docker set (options, search).
|
||||
{ printf 'nameserver 127.0.0.1\n'
|
||||
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
|
||||
} > "$STATE/resolv.jailed" 2>/dev/null
|
||||
[ -s "$STATE/resolv.jailed" ] || return 0
|
||||
cat "$STATE/resolv.jailed" > /etc/resolv.conf
|
||||
}
|
||||
|
||||
dnsjail_apply || true
|
||||
@@ -0,0 +1,263 @@
|
||||
#!/bin/bash
|
||||
# Read the harness registry and derive per-harness credentials from it.
|
||||
#
|
||||
# Source it — the whole point is exporting into the caller's environment, which a subshell
|
||||
# would lose:
|
||||
#
|
||||
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
|
||||
# harness_setup_credentials
|
||||
#
|
||||
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
|
||||
# re-derives and rewrites the auth files before an interactive launch; and
|
||||
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
|
||||
# on top.
|
||||
#
|
||||
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
|
||||
# post-creates run with -e). An unguarded failure below therefore aborts container
|
||||
# creation, which is why every failure site is individually guarded rather than relying on
|
||||
# this line.
|
||||
set -uo pipefail
|
||||
|
||||
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
|
||||
|
||||
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
|
||||
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
|
||||
# the first one that can actually import it rather than assuming.
|
||||
_raccoon_python() {
|
||||
local p
|
||||
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
|
||||
[ -n "$p" ] || continue
|
||||
command -v "$p" >/dev/null 2>&1 || continue
|
||||
if "$p" -c "import tomllib" >/dev/null 2>&1; then
|
||||
printf '%s' "$p"
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
_harness_query() {
|
||||
local py
|
||||
py=$(_raccoon_python) || return 1
|
||||
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
|
||||
}
|
||||
|
||||
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
|
||||
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
|
||||
# whitespace anywhere, so deleting rather than trimming needs no cases.
|
||||
_harness_trim() {
|
||||
local out
|
||||
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
|
||||
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
|
||||
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
|
||||
printf '%s' "${out:-$1}"
|
||||
}
|
||||
|
||||
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
|
||||
_harness_proxy_root() {
|
||||
local base_url
|
||||
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
|
||||
[ -n "$base_url" ] || return 1
|
||||
base_url="${base_url%"${base_url##*[!/]}"}"
|
||||
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
|
||||
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
|
||||
# URL that is a bare host with no path — a provider's own API root rather than the
|
||||
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
|
||||
case "${base_url#*://}" in
|
||||
*/*) printf '%s' "${base_url%/*}" ;;
|
||||
*) return 2 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
harness_setup_credentials() {
|
||||
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
|
||||
# note at the top), and a bare failing assignment would exit the caller's post-create
|
||||
# outright — silently, since the failure paths below are what do the explaining.
|
||||
local root rc=0
|
||||
root="$(_harness_proxy_root)" || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
if [ "$rc" -eq 2 ]; then
|
||||
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
|
||||
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
|
||||
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
|
||||
echo "harness-setup: authenticated. Use the base URL you were given." >&2
|
||||
else
|
||||
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
|
||||
export ANTHROPIC_BASE_URL
|
||||
local key
|
||||
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
|
||||
if [ -z "$key" ]; then
|
||||
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
|
||||
return 0
|
||||
fi
|
||||
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
|
||||
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
|
||||
export ANTHROPIC_API_KEY="$key"
|
||||
|
||||
local id key_env base_url_env proxy_path
|
||||
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
|
||||
[ -n "$key_env" ] || continue
|
||||
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
|
||||
if [ -z "${!key_env:-}" ]; then
|
||||
export "$key_env=$key"
|
||||
fi
|
||||
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
|
||||
export "$base_url_env=$root/$proxy_path"
|
||||
fi
|
||||
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
|
||||
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
|
||||
harness_write_auth() {
|
||||
local id auth_path key_env target key py
|
||||
py=$(_raccoon_python) || {
|
||||
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
|
||||
return 0
|
||||
}
|
||||
while IFS=$'\t' read -r id auth_path key_env; do
|
||||
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
|
||||
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
|
||||
# too — this is the value that reaches the file the harness authenticates with.
|
||||
key="$(_harness_trim "${!key_env:-}")"
|
||||
if [ -z "$key" ]; then
|
||||
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
|
||||
continue
|
||||
fi
|
||||
target=$(eval "printf '%s' \"$auth_path\"") || {
|
||||
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
|
||||
continue
|
||||
}
|
||||
mkdir -p "$(dirname "$target")" || {
|
||||
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
|
||||
continue
|
||||
}
|
||||
# json.dumps, not printf: a key containing a quote or backslash would otherwise
|
||||
# produce a file the CLI cannot parse, and the failure would surface as an auth
|
||||
# error rather than a malformed file.
|
||||
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
|
||||
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
|
||||
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
|
||||
"$py" -c 'import json, os
|
||||
target = os.environ["RACCOON_AUTH_TARGET"]
|
||||
tmp = target + ".raccoon-tmp." + str(os.getpid())
|
||||
try:
|
||||
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
|
||||
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
|
||||
fh.write("\n")
|
||||
os.replace(tmp, target)
|
||||
except OSError:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
raise SystemExit(1)
|
||||
'; then
|
||||
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
|
||||
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
|
||||
continue
|
||||
fi
|
||||
echo "harness-setup: $id auth -> $target" >&2
|
||||
done < <(_harness_query --auth-files 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
|
||||
# leaving every other line — the explore surface's [hooks] table included — untouched.
|
||||
harness_refresh_config_keys() {
|
||||
local id config_path blob target py
|
||||
py=$(_raccoon_python) || return 0
|
||||
# The surface only decides what a CREATE writes. An update takes the root keys off the
|
||||
# front of the same blob, so a surface's tables survive byte-for-byte either way.
|
||||
while IFS=$'\t' read -r id config_path blob; do
|
||||
[ -n "$config_path" ] && [ -n "$blob" ] || continue
|
||||
target=$(eval "printf '%s' \"$config_path\"") || continue
|
||||
mkdir -p "$(dirname "$target")" || continue
|
||||
if printf '%s' "$blob" | base64 -d |
|
||||
RACCOON_CONFIG_TARGET="$target" "$py" -c '
|
||||
import os, re, sys, tomllib
|
||||
|
||||
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
|
||||
|
||||
target = os.environ["RACCOON_CONFIG_TARGET"]
|
||||
text = sys.stdin.read()
|
||||
# Empty counts as unresolved: writing an empty base URL would break a container whose
|
||||
# config is currently right, which is the one thing this must never do.
|
||||
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
|
||||
raise SystemExit(1)
|
||||
text = os.path.expandvars(text)
|
||||
|
||||
wanted = []
|
||||
for line in text.splitlines():
|
||||
if line.lstrip().startswith("["):
|
||||
break
|
||||
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
|
||||
if m:
|
||||
wanted.append((m.group(1), line.rstrip()))
|
||||
if not wanted:
|
||||
raise SystemExit(0)
|
||||
|
||||
mode = None
|
||||
if os.path.exists(target):
|
||||
try:
|
||||
with open(target, encoding="utf-8") as fh:
|
||||
lines = fh.read().splitlines()
|
||||
mode = os.stat(target).st_mode & 0o777
|
||||
except OSError:
|
||||
raise SystemExit(1)
|
||||
# Everything from the first table header on belongs to a table. A key appended after
|
||||
# one is reparented into it, so both the search and the insert stay above the line.
|
||||
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
|
||||
changed = False
|
||||
for key, line in wanted:
|
||||
# The quoted spelling is the same key: replacing it beats adding a duplicate.
|
||||
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
|
||||
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
|
||||
if at is None:
|
||||
if root_end < len(lines) and lines[root_end].strip():
|
||||
lines.insert(root_end, "")
|
||||
lines.insert(root_end, line)
|
||||
root_end += 1
|
||||
changed = True
|
||||
elif lines[at] != line:
|
||||
lines[at] = line
|
||||
changed = True
|
||||
if not changed:
|
||||
raise SystemExit(0)
|
||||
out = "\n".join(lines).rstrip("\n") + "\n"
|
||||
else:
|
||||
# No file means container-create could not write one, so write what it would have:
|
||||
# on the explore surface that is the capture hooks too, not just the root keys.
|
||||
out = HEADER + "\n" + text
|
||||
|
||||
try:
|
||||
doc = tomllib.loads(out)
|
||||
except tomllib.TOMLDecodeError:
|
||||
raise SystemExit(1)
|
||||
# Parsing is not enough: a line edit can land inside a multi-line value, which still
|
||||
# parses while leaving the key unset. Require every key to have reached the root.
|
||||
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
|
||||
raise SystemExit(1)
|
||||
|
||||
# Pid-suffixed: two launches at once must not write the same scratch path.
|
||||
tmp = target + ".raccoon-tmp." + str(os.getpid())
|
||||
try:
|
||||
with open(tmp, "w", encoding="utf-8") as fh:
|
||||
fh.write(out)
|
||||
if mode is not None:
|
||||
os.chmod(tmp, mode)
|
||||
os.replace(tmp, target)
|
||||
except OSError:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
raise SystemExit(1)
|
||||
'; then
|
||||
echo "harness-setup: $id config keys refreshed -> $target" >&2
|
||||
fi
|
||||
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
|
||||
}
|
||||
@@ -0,0 +1,418 @@
|
||||
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
|
||||
|
||||
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
|
||||
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
|
||||
package.json has no zod/smol-toml).
|
||||
|
||||
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
|
||||
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
|
||||
copies.
|
||||
|
||||
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
|
||||
sandbox agent, a plain unit test, or the devcontainer python alike.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import tomllib
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
|
||||
|
||||
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Harness:
|
||||
"""One harness, as declared in harness-registry.toml."""
|
||||
|
||||
id: str
|
||||
label: str
|
||||
agent_import_path: str
|
||||
model_id_shape: str
|
||||
writes_atif: bool
|
||||
capture: bool
|
||||
seed_native: bool
|
||||
seed_atif: bool
|
||||
agent_import_path_single_turn: str | None = None
|
||||
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
|
||||
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
|
||||
# has no `Read` equivalent to switch toolsets for.
|
||||
agent_import_path_browser: str | None = None
|
||||
agent_import_path_single_turn_browser: str | None = None
|
||||
import_path_aliases: tuple[str, ...] = ()
|
||||
legacy_bare_model_rows: bool = False
|
||||
default_model: str | None = None
|
||||
effort_kwarg: str = ""
|
||||
effort_default: str | None = None
|
||||
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
|
||||
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
|
||||
fast_kwarg: str = ""
|
||||
key_env: str | None = None
|
||||
base_url_env: str | None = None
|
||||
proxy_path: str | None = None
|
||||
flaky_hangs: bool = False
|
||||
enabled: bool = True
|
||||
# Worker-container fields; see the registry header.
|
||||
authoring: bool = False
|
||||
cli: str | None = None
|
||||
install: str | None = None
|
||||
skills_dir: str | None = None
|
||||
auth_path: str | None = None
|
||||
auth_key_env: str | None = None
|
||||
explore_launch: str | None = None
|
||||
config_path: str | None = None
|
||||
# Config the harness needs wherever it runs, trial sandbox included.
|
||||
agent_config: str | None = None
|
||||
# Config for both worker containers (explore and authoring).
|
||||
container_config: str | None = None
|
||||
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
|
||||
# explore/plugins/. Writing them in authoring would register hooks against files that
|
||||
# are not there, firing on every prompt.
|
||||
explore_config: str | None = None
|
||||
# Fields added for a later phase, kept verbatim so this loader doesn't have to
|
||||
# be edited in lockstep with the schema.
|
||||
extra: dict = field(default_factory=dict, compare=False)
|
||||
|
||||
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
|
||||
"""Agent class to launch. Multi-turn tasks need the resuming class; a
|
||||
single-turn task given it would try to resume a session that isn't there.
|
||||
|
||||
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
|
||||
built-in — a different toolset is a different agent, so it is a different class with
|
||||
its own name rather than a flag on the canonical one. Harnesses without a variant fall
|
||||
through to their normal class."""
|
||||
if browser:
|
||||
variant = (
|
||||
self.agent_import_path_browser
|
||||
if multi_turn
|
||||
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
|
||||
)
|
||||
if variant:
|
||||
return variant
|
||||
if multi_turn:
|
||||
return self.agent_import_path
|
||||
return self.agent_import_path_single_turn or self.agent_import_path
|
||||
|
||||
def row_label(self, model: str) -> str:
|
||||
"""Row identity for one trial: bare model for legacy harnesses (so
|
||||
published manifests keep their labels), else ``<harness>:<model>``."""
|
||||
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
|
||||
|
||||
def agent_config_overrides(self) -> dict[str, str]:
|
||||
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
|
||||
|
||||
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
|
||||
interpolates these into a shell command, which would strip the quotes anyway;
|
||||
emitting them would only make the result depend on how many shell layers the
|
||||
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
|
||||
binary as ``model=o3``).
|
||||
|
||||
These settings ride the command line as ``-c dotted.key=value`` everywhere the
|
||||
harness runs, never a config file. A trial sandbox rules the file out: the
|
||||
harness's own runner appends root keys to it, and TOML has no way back to the
|
||||
root scope once a table has opened, so a table we appended would swallow them.
|
||||
Overrides compose in any order and beat the file, so the same rendering serves
|
||||
the explore launcher too — one declaration, one mechanism.
|
||||
"""
|
||||
if not self.agent_config:
|
||||
return {}
|
||||
try:
|
||||
parsed = tomllib.loads(self.agent_config)
|
||||
except tomllib.TOMLDecodeError as exc:
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config is not valid TOML ({exc})"
|
||||
) from exc
|
||||
|
||||
flat: dict[str, str] = {}
|
||||
|
||||
def walk(node: dict, prefix: str) -> None:
|
||||
for key, value in node.items():
|
||||
path = f"{prefix}{key}"
|
||||
if isinstance(value, dict):
|
||||
walk(value, f"{path}.")
|
||||
elif isinstance(value, bool):
|
||||
flat[path] = "true" if value else "false"
|
||||
elif isinstance(value, (int, float)):
|
||||
flat[path] = str(value)
|
||||
elif isinstance(value, str):
|
||||
if value != value.strip() or any(c in value for c in " \"'\\"):
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config key {path!r} has a value needing "
|
||||
"shell quoting, which the -c override form cannot carry"
|
||||
)
|
||||
flat[path] = value
|
||||
else:
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config key {path!r} has type "
|
||||
f"{type(value).__name__}, which has no -c override form"
|
||||
)
|
||||
|
||||
walk(parsed, "")
|
||||
return flat
|
||||
|
||||
def container_config_text(self, *, surface: str) -> str | None:
|
||||
"""Config file body for a worker container. `surface` is "explore" or
|
||||
"authoring"; explore additionally gets `explore_config`. Root keys come from
|
||||
`container_config` first, so appending a table section stays valid TOML."""
|
||||
parts = [self.container_config]
|
||||
if surface == "explore":
|
||||
parts.append(self.explore_config)
|
||||
kept = [part.strip("\n") for part in parts if part and part.strip()]
|
||||
return "\n\n".join(kept) + "\n" if kept else None
|
||||
|
||||
def agent_config_flags(self) -> str:
|
||||
"""``agent_config`` as a ``-c key=value`` command-line string."""
|
||||
return " ".join(
|
||||
f"-c {key}={value}"
|
||||
for key, value in sorted(self.agent_config_overrides().items())
|
||||
)
|
||||
|
||||
def explore_launch_command(self) -> str | None:
|
||||
"""``explore_launch`` with the registry's own values substituted in.
|
||||
|
||||
The worker's Explore session and the trial must run the same agent, so the
|
||||
model, effort and reductions are declared once here and rendered into both.
|
||||
A literal in the launch string would be a second declaration, and the two
|
||||
would drift the first time one of them was updated alone.
|
||||
|
||||
Only these three placeholders are substituted; ``$@`` and
|
||||
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
|
||||
"""
|
||||
if not self.explore_launch:
|
||||
return None
|
||||
return (
|
||||
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
|
||||
.replace("$RACCOON_MODEL", self.default_model or "")
|
||||
.replace("$RACCOON_EFFORT", self.effort_default or "")
|
||||
)
|
||||
|
||||
def known_import_paths(self) -> tuple[str, ...]:
|
||||
paths = [self.agent_import_path, *self.import_path_aliases]
|
||||
if self.agent_import_path_single_turn:
|
||||
paths.append(self.agent_import_path_single_turn)
|
||||
return tuple(paths)
|
||||
|
||||
|
||||
_KNOWN_FIELDS = frozenset(
|
||||
{
|
||||
"id",
|
||||
"label",
|
||||
"agent_import_path",
|
||||
"agent_import_path_single_turn",
|
||||
"agent_import_path_browser",
|
||||
"agent_import_path_single_turn_browser",
|
||||
"import_path_aliases",
|
||||
"legacy_bare_model_rows",
|
||||
"default_model",
|
||||
"model_id_shape",
|
||||
"effort_kwarg",
|
||||
"effort_default",
|
||||
"fast_kwarg",
|
||||
"key_env",
|
||||
"base_url_env",
|
||||
"proxy_path",
|
||||
"writes_atif",
|
||||
"capture",
|
||||
"seed_native",
|
||||
"seed_atif",
|
||||
"flaky_hangs",
|
||||
"enabled",
|
||||
"authoring",
|
||||
"cli",
|
||||
"install",
|
||||
"skills_dir",
|
||||
"auth_path",
|
||||
"auth_key_env",
|
||||
"explore_launch",
|
||||
"config_path",
|
||||
"agent_config",
|
||||
"container_config",
|
||||
"explore_config",
|
||||
}
|
||||
)
|
||||
|
||||
_REQUIRED_FIELDS = (
|
||||
"id",
|
||||
"label",
|
||||
"agent_import_path",
|
||||
"model_id_shape",
|
||||
"writes_atif",
|
||||
"capture",
|
||||
"seed_native",
|
||||
"seed_atif",
|
||||
)
|
||||
|
||||
|
||||
class HarnessRegistryError(ValueError):
|
||||
"""Malformed registry. Raised rather than tolerated: a broken registry is a
|
||||
broken deployment, and silently defaulting would pick the wrong agent."""
|
||||
|
||||
|
||||
def _references_agent_flags(launch: str) -> bool:
|
||||
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HarnessRegistry:
|
||||
version: int
|
||||
harnesses: tuple[Harness, ...]
|
||||
|
||||
def all(self) -> tuple[Harness, ...]:
|
||||
return self.harnesses
|
||||
|
||||
def enabled(self) -> tuple[Harness, ...]:
|
||||
return tuple(h for h in self.harnesses if h.enabled)
|
||||
|
||||
def authoring(self) -> tuple[Harness, ...]:
|
||||
"""Harnesses a worker can author with — what the worker containers install.
|
||||
Narrower than enabled(): a harness can be runnable in a trial without having
|
||||
an authoring story (no CLI to converse with, or no capture)."""
|
||||
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
|
||||
|
||||
def find(self, harness_id: str) -> Harness | None:
|
||||
return next((h for h in self.harnesses if h.id == harness_id), None)
|
||||
|
||||
def require(self, harness_id: str) -> Harness:
|
||||
harness = self.find(harness_id)
|
||||
if harness is not None:
|
||||
return harness
|
||||
available = ", ".join(sorted(h.id for h in self.enabled()))
|
||||
raise HarnessRegistryError(
|
||||
f'Unknown harness "{harness_id}". Available: {available}'
|
||||
)
|
||||
|
||||
def by_import_path(self, agent: str) -> Harness | None:
|
||||
"""Resolve an agent identity — a ``name()`` or import path from
|
||||
``result.json`` ``config.agent``, or a manifest row — to its harness."""
|
||||
needle = (agent or "").strip()
|
||||
if not needle:
|
||||
return None
|
||||
for harness in self.harnesses:
|
||||
if needle == harness.id or needle in harness.known_import_paths():
|
||||
return harness
|
||||
return None
|
||||
|
||||
|
||||
def _build(entry: dict, index: int) -> Harness:
|
||||
for name in _REQUIRED_FIELDS:
|
||||
if name not in entry:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}]: missing required field '{name}'"
|
||||
)
|
||||
shape = entry["model_id_shape"]
|
||||
if shape not in MODEL_ID_SHAPES:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
|
||||
f"{sorted(MODEL_ID_SHAPES)}"
|
||||
)
|
||||
# These three reach `eval` in setup-harnesses.sh, which is how they support the
|
||||
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
|
||||
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
|
||||
# is ours, but "ours" is not an argument that survives a careless future edit.
|
||||
for shell_field in ("config_path", "auth_path", "skills_dir"):
|
||||
value = entry.get(shell_field)
|
||||
if not isinstance(value, str):
|
||||
continue
|
||||
# A backtick or $( executes outright. A double quote closes the string these are
|
||||
# interpolated into, and a semicolon then starts a new command inside it — same
|
||||
# outcome, one step removed.
|
||||
bad = [t for t in ("`", "$(", '"', ";") if t in value]
|
||||
if bad:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): {shell_field} contains "
|
||||
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
|
||||
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
|
||||
)
|
||||
|
||||
launch = entry.get("explore_launch")
|
||||
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): declares agent_config but its "
|
||||
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
|
||||
"would then run with a different toolset than the trial it is authoring "
|
||||
"for, which is the drift agent_config exists to prevent."
|
||||
)
|
||||
return Harness(
|
||||
id=entry["id"],
|
||||
label=entry["label"],
|
||||
agent_import_path=entry["agent_import_path"],
|
||||
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
|
||||
agent_import_path_browser=entry.get("agent_import_path_browser"),
|
||||
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
|
||||
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
|
||||
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
|
||||
default_model=entry.get("default_model"),
|
||||
model_id_shape=shape,
|
||||
effort_kwarg=entry.get("effort_kwarg", ""),
|
||||
effort_default=entry.get("effort_default"),
|
||||
fast_kwarg=entry.get("fast_kwarg", ""),
|
||||
key_env=entry.get("key_env"),
|
||||
base_url_env=entry.get("base_url_env"),
|
||||
proxy_path=entry.get("proxy_path"),
|
||||
writes_atif=bool(entry["writes_atif"]),
|
||||
capture=bool(entry["capture"]),
|
||||
seed_native=bool(entry["seed_native"]),
|
||||
seed_atif=bool(entry["seed_atif"]),
|
||||
flaky_hangs=bool(entry.get("flaky_hangs", False)),
|
||||
enabled=bool(entry.get("enabled", True)),
|
||||
authoring=bool(entry.get("authoring", False)),
|
||||
cli=entry.get("cli"),
|
||||
install=entry.get("install"),
|
||||
skills_dir=entry.get("skills_dir"),
|
||||
auth_path=entry.get("auth_path"),
|
||||
auth_key_env=entry.get("auth_key_env"),
|
||||
explore_launch=entry.get("explore_launch"),
|
||||
config_path=entry.get("config_path"),
|
||||
agent_config=entry.get("agent_config"),
|
||||
container_config=entry.get("container_config"),
|
||||
explore_config=entry.get("explore_config"),
|
||||
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
|
||||
)
|
||||
|
||||
|
||||
_cache: dict[Path, HarnessRegistry] = {}
|
||||
|
||||
|
||||
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
|
||||
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
|
||||
file, a duplicate id, or an import path claimed by two harnesses (which would
|
||||
make ``by_import_path`` depend on declaration order)."""
|
||||
resolved = Path(path).resolve()
|
||||
if resolved in _cache:
|
||||
return _cache[resolved]
|
||||
|
||||
with open(resolved, "rb") as handle:
|
||||
doc = tomllib.load(handle)
|
||||
|
||||
if "version" not in doc:
|
||||
raise HarnessRegistryError("harness-registry: missing 'version'")
|
||||
entries = doc.get("harness") or []
|
||||
if not entries:
|
||||
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
|
||||
|
||||
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
|
||||
|
||||
seen_ids: set[str] = set()
|
||||
for harness in harnesses:
|
||||
if harness.id in seen_ids:
|
||||
raise HarnessRegistryError(
|
||||
f"harness-registry: duplicate harness id: {harness.id}"
|
||||
)
|
||||
seen_ids.add(harness.id)
|
||||
|
||||
owners: dict[str, str] = {}
|
||||
for harness in harnesses:
|
||||
for import_path in harness.known_import_paths():
|
||||
owner = owners.get(import_path)
|
||||
if owner is not None and owner != harness.id:
|
||||
raise HarnessRegistryError(
|
||||
f'harness-registry: import path "{import_path}" claimed by both '
|
||||
f'"{owner}" and "{harness.id}"'
|
||||
)
|
||||
owners[import_path] = harness.id
|
||||
|
||||
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
|
||||
_cache[resolved] = registry
|
||||
return registry
|
||||
@@ -0,0 +1,324 @@
|
||||
/**
|
||||
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
|
||||
* that reference runs and detector reports depend on.
|
||||
*
|
||||
* A reference run is only meaningful for the task inputs it actually ran
|
||||
* against: the prompt (instruction.md), the snapshot session
|
||||
* (environment/session.jsonl), the workspace patch
|
||||
* (environment/workspace.patch), and the gitref the workspace is built from
|
||||
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
|
||||
* revision of instruction.md + the task's holistic rubric
|
||||
* (tests/holistic-rubric.md on current tasks; tests/grader-guidance-consolidated.md
|
||||
* on tasks created before the rename) — and, for the rubric detectors, the
|
||||
* task's atomic rubric (tests/atomic-rubric.yaml; tests/rubrics.yaml on tasks
|
||||
* converted before the rename) and tests/grader-context.md. When any of those
|
||||
* change after the artifact was produced, the artifact is stale — it describes
|
||||
* an older revision of the task than the one being packaged.
|
||||
*
|
||||
* This module is the single source of truth for WHAT gets checksummed and how
|
||||
* captures are compared. Capture sites (copy-reference-run.ts,
|
||||
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
|
||||
* the artifact; submit-task.ts re-captures at packaging time and diffs.
|
||||
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
|
||||
* mask an mtime, but can't change a sha256.
|
||||
*
|
||||
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
|
||||
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
|
||||
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
|
||||
* occupies the same relative path there (mirroring the check-devcontainer
|
||||
* pattern).
|
||||
*
|
||||
* Related but deliberately separate: `computeDeliveryHash` (repo-side
|
||||
* delivery script — grep the internal repo for it; not shipped with the
|
||||
* toolkit) hashes an overlapping input set for delivery idempotency. It is
|
||||
* NOT built on this module because its hash format is load-bearing (a
|
||||
* changed hash re-delivers every task); if you change WHAT counts as a task
|
||||
* input here, check whether the delivery hash needs the same change.
|
||||
*/
|
||||
|
||||
import { createHash } from 'node:crypto';
|
||||
import { existsSync, readFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
|
||||
/** Bump when the record shape changes incompatibly. */
|
||||
export const INPUT_CHECKSUMS_VERSION = 1;
|
||||
|
||||
/** Filename of the record inside a reference-run directory. */
|
||||
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
|
||||
|
||||
/**
|
||||
* Where in the artifact lifecycle a capture happened. The moment matters for
|
||||
* how much a "fresh" verdict can be trusted:
|
||||
*
|
||||
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
|
||||
* The strongest evidence: the record is what the agent ran
|
||||
* against, whatever got edited afterwards.
|
||||
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
|
||||
* no run-time stamp. An input edited between harbor-run and the
|
||||
* copy is recorded at its post-edit state, so a stale run can
|
||||
* read fresh.
|
||||
* - 'stamp' — right after a detector report is written
|
||||
* (record-detector-inputs.ts in a worker checkout; the
|
||||
* internal repo's detector save path stamps the same way).
|
||||
* - 'mirror' — retired: written by the repo-side flow that re-materialized
|
||||
* canonical detector reports to disk back when reports had a
|
||||
* remote canonical store. Reports are local-only now, so no
|
||||
* current code writes it; the member stays so old stamps keep
|
||||
* their recorded method when read.
|
||||
* - 'regrade' — a re-grade of an existing run. Present on records already on
|
||||
* disk; no current code path writes it.
|
||||
*
|
||||
* Absent on records written before this field existed.
|
||||
*/
|
||||
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror' | 'regrade';
|
||||
|
||||
/** Runtime mirror of {@link TaskInputCaptureMethod}, for validating a record read from disk. */
|
||||
export const TASK_INPUT_CAPTURE_METHODS = Object.freeze([
|
||||
'run',
|
||||
'copy',
|
||||
'stamp',
|
||||
'mirror',
|
||||
'regrade',
|
||||
] as const satisfies readonly TaskInputCaptureMethod[]);
|
||||
|
||||
/**
|
||||
* The checksums of a task's inputs as they stood at capture time. Every hash
|
||||
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
|
||||
* that later becomes a hash — or vice versa — is a change like any other).
|
||||
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
|
||||
* than hashed so a mismatch message can show it.
|
||||
*/
|
||||
export interface TaskInputChecksums {
|
||||
readonly version: number;
|
||||
/** ISO-8601 timestamp of the capture. */
|
||||
readonly capturedAt: string;
|
||||
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
|
||||
readonly capturedBy?: TaskInputCaptureMethod;
|
||||
/**
|
||||
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
|
||||
* by launch-time captures so stamping can be scoped to the right task's
|
||||
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
|
||||
* stamp is detectable after the fact.
|
||||
*/
|
||||
readonly taskSlug?: string;
|
||||
/**
|
||||
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
|
||||
* after the harbor-path scrub deliberately rewrote those docs
|
||||
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
|
||||
* task; only the doc bytes were normalized.
|
||||
*/
|
||||
readonly restampedAt?: string;
|
||||
readonly inputs: {
|
||||
readonly prompt: string | null;
|
||||
readonly graderGuidance: string | null;
|
||||
readonly sessionJsonl: string | null;
|
||||
readonly workspacePatch: string | null;
|
||||
readonly gitref: string | null;
|
||||
/**
|
||||
* tests/grader-guidance-consolidated.md — the holistic rubric under its
|
||||
* pre-rename filename, which every task created before the rename keeps;
|
||||
* hashed when present, null otherwise. Absent (undefined) on records
|
||||
* captured before the field existed; comparisons skip a field the record
|
||||
* predates, so old captures stay fresh until they are re-stamped.
|
||||
*/
|
||||
readonly graderGuidanceConsolidated?: string | null;
|
||||
/**
|
||||
* tests/holistic-rubric.md — the holistic rubric under its current
|
||||
* filename (each task carries exactly one of this and the pre-rename
|
||||
* name above). Hashed when present, null otherwise. Absent (undefined)
|
||||
* on records captured before the field existed; comparisons skip a
|
||||
* field the record predates. The task-checksum digest serializes fields
|
||||
* in this order and appends new fields at the end (see
|
||||
* scripts/lib/grader-run-checksums.ts).
|
||||
*/
|
||||
readonly holisticRubric?: string | null;
|
||||
/**
|
||||
* tests/atomic-rubric.yaml — the task's atomic rubric under its current
|
||||
* filename. Hashed when present, null otherwise. Absent (undefined) on
|
||||
* records captured before the field existed; comparisons skip a field
|
||||
* the record predates.
|
||||
*/
|
||||
readonly atomicRubric?: string | null;
|
||||
/**
|
||||
* tests/rubrics.yaml — the atomic rubric under its pre-rename filename,
|
||||
* which tasks converted before the rename keep. Hashed when present,
|
||||
* null otherwise; absent (undefined) on records captured before the
|
||||
* field existed.
|
||||
*/
|
||||
readonly rubricsYaml?: string | null;
|
||||
/**
|
||||
* tests/grader-context.md — the context document the rubric grader modes
|
||||
* read beside the atomic rubric. Hashed when present, null otherwise;
|
||||
* absent (undefined) on records captured before the field existed. Last
|
||||
* in field order per the append-at-the-end digest rule above.
|
||||
*/
|
||||
readonly graderContext?: string | null;
|
||||
};
|
||||
}
|
||||
|
||||
export type TaskInputName = keyof TaskInputChecksums['inputs'];
|
||||
|
||||
/** Human-readable component names, used verbatim in staleness warnings. Frozen:
|
||||
* its key set is the runtime source of truth for the task-input axes. */
|
||||
export const INPUT_LABELS = Object.freeze({
|
||||
prompt: 'prompt (instruction.md)',
|
||||
graderGuidance: 'legacy-era grader guidance (tests/grader-guidance.md)',
|
||||
graderGuidanceConsolidated: 'holistic rubric (tests/grader-guidance-consolidated.md)',
|
||||
sessionJsonl: 'session snapshot (environment/session.jsonl)',
|
||||
workspacePatch: 'workspace patch (environment/workspace.patch)',
|
||||
gitref: 'gitref (task.toml commit)',
|
||||
holisticRubric: 'holistic rubric (tests/holistic-rubric.md)',
|
||||
atomicRubric: 'atomic rubric (tests/atomic-rubric.yaml)',
|
||||
rubricsYaml: 'atomic rubric (tests/rubrics.yaml)',
|
||||
graderContext: 'grader context (tests/grader-context.md)',
|
||||
} as const satisfies Record<TaskInputName, string>);
|
||||
|
||||
/**
|
||||
* The inputs that shape what the AGENT saw and did. Changing any of them means
|
||||
* a captured reference run no longer reflects the task being packaged, and
|
||||
* only re-running the agent can fix that. The holistic-rubric files are
|
||||
* deliberately NOT in this set: editing the rubric stales the run's GRADE,
|
||||
* not the run itself, and `scripts/harbor-regrade` re-derives grades without
|
||||
* re-running the agent.
|
||||
*/
|
||||
export const REFERENCE_RUN_INPUTS = Object.freeze([
|
||||
'prompt',
|
||||
'sessionJsonl',
|
||||
'workspacePatch',
|
||||
'gitref',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/**
|
||||
* The inputs a detector report assesses — instruction.md plus whichever
|
||||
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
|
||||
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
|
||||
* before the rename, plus the legacy-era plain-named file when a task
|
||||
* authored on an earlier generation carries one), plus the atomic-rubric
|
||||
* files the rubric detectors assess (tests/atomic-rubric.yaml, the pre-rename
|
||||
* tests/rubrics.yaml, and tests/grader-context.md). An absent file hashes to
|
||||
* null on both sides and never diffs. Compared by content.
|
||||
*/
|
||||
export const DETECTOR_REPORT_INPUTS = Object.freeze([
|
||||
'prompt',
|
||||
'graderGuidance',
|
||||
'graderGuidanceConsolidated',
|
||||
'holisticRubric',
|
||||
'atomicRubric',
|
||||
'rubricsYaml',
|
||||
'graderContext',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
|
||||
function sha256File(filePath: string): string | null {
|
||||
if (!existsSync(filePath)) return null;
|
||||
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
|
||||
}
|
||||
|
||||
/**
|
||||
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
|
||||
* missing, unreadable, or has no commit line — a read failure downgrades to
|
||||
* "absent" rather than crashing a capture or a validation sweep.
|
||||
*
|
||||
* Deliberately a regex, not a TOML parser: this module ships in the worker
|
||||
* toolkit, where a new runtime dep would break packaging for every worker
|
||||
* whose container predates the dep (npm install runs only on container
|
||||
* create, and containers survive toolkit upgrades). build-workspace.sh reads
|
||||
* the same key with the same grep-a-`commit`-line approach. The one `commit`
|
||||
* key in a task.toml is `[metadata].commit`, so anchoring to the first
|
||||
* `commit = "…"` line is exact in practice.
|
||||
*/
|
||||
function readGitref(taskDir: string): string | null {
|
||||
const tomlPath = join(taskDir, 'task.toml');
|
||||
if (!existsSync(tomlPath)) return null;
|
||||
try {
|
||||
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
|
||||
readFileSync(tomlPath, 'utf-8')
|
||||
);
|
||||
const commit = match?.[1] ?? match?.[2];
|
||||
return commit && commit.length > 0 ? commit : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Checksum the task inputs as they stand right now under `taskDir`. */
|
||||
export function captureTaskInputs(
|
||||
taskDir: string,
|
||||
capturedBy?: TaskInputCaptureMethod
|
||||
): TaskInputChecksums {
|
||||
return Object.freeze({
|
||||
version: INPUT_CHECKSUMS_VERSION,
|
||||
capturedAt: new Date().toISOString(),
|
||||
...(capturedBy ? { capturedBy } : {}),
|
||||
inputs: Object.freeze({
|
||||
prompt: sha256File(join(taskDir, 'instruction.md')),
|
||||
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
|
||||
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
|
||||
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
|
||||
gitref: readGitref(taskDir),
|
||||
graderGuidanceConsolidated: sha256File(
|
||||
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
|
||||
),
|
||||
holisticRubric: sha256File(join(taskDir, 'tests', 'holistic-rubric.md')),
|
||||
atomicRubric: sha256File(join(taskDir, 'tests', 'atomic-rubric.yaml')),
|
||||
rubricsYaml: sha256File(join(taskDir, 'tests', 'rubrics.yaml')),
|
||||
graderContext: sha256File(join(taskDir, 'tests', 'grader-context.md')),
|
||||
}),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a previously captured record. Returns null when the file is missing or
|
||||
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
|
||||
* future incompatible version) — callers treat null as "staleness unknowable",
|
||||
* never as an error. No zod here: this module ships in the worker toolkit,
|
||||
* whose dependency set stays minimal, so the guard is manual.
|
||||
*/
|
||||
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
|
||||
if (!existsSync(filePath)) return null;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (typeof parsed !== 'object' || parsed === null) return null;
|
||||
const record = parsed as TaskInputChecksums;
|
||||
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
|
||||
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
|
||||
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
|
||||
const value = record.inputs[name];
|
||||
// undefined = the record predates this input field; still a valid capture.
|
||||
if (value !== undefined && value !== null && typeof value !== 'string') return null;
|
||||
}
|
||||
// An unrecognized capture method is dropped, not rejected: the field is
|
||||
// provenance colour, and rejecting would flip the whole run to "unknowable".
|
||||
const method: unknown = record.capturedBy;
|
||||
const isKnown = TASK_INPUT_CAPTURE_METHODS.includes(method as TaskInputCaptureMethod);
|
||||
if (method !== undefined && !isKnown) {
|
||||
const { capturedBy: _dropped, ...rest } = record;
|
||||
return rest;
|
||||
}
|
||||
return record;
|
||||
}
|
||||
|
||||
/**
|
||||
* Which of `names` changed between a recorded capture and the current state?
|
||||
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
|
||||
* whose value differs — including absent→present and present→absent flips. A
|
||||
* field the recorded capture predates (the key is not in the record at all)
|
||||
* is skipped: freshness on that axis is unknowable, and flagging every old
|
||||
* record the moment a new axis ships would drown the real signal. (The
|
||||
* task-checksum fold folds absence as 'null' instead — it only ever reads a
|
||||
* fresh capture, which is total, so the two never disagree in practice.)
|
||||
*/
|
||||
export function diffTaskInputs(
|
||||
recorded: TaskInputChecksums,
|
||||
current: TaskInputChecksums,
|
||||
names: readonly TaskInputName[]
|
||||
): string[] {
|
||||
return names
|
||||
.filter((name) => name in recorded.inputs)
|
||||
.filter((name) => recorded.inputs[name] !== current.inputs[name])
|
||||
.map((name) => INPUT_LABELS[name]);
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
/**
|
||||
* Wrap a notice in a banner loud enough to survive a scrollback.
|
||||
*
|
||||
* Yellow only when stderr is a terminal, so piped logs stay clean.
|
||||
*/
|
||||
export function banner(message: string, headline: string): string {
|
||||
const RULE = '#'.repeat(78);
|
||||
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
|
||||
|
||||
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
|
||||
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
|
||||
return `${color[0]}${body}${color[1]}`;
|
||||
}
|
||||
@@ -0,0 +1,431 @@
|
||||
/**
|
||||
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
|
||||
*
|
||||
* `environment/Dockerfile`, `tests/test.sh`, and
|
||||
* `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
|
||||
* are the same in every task: they decide how the
|
||||
* trial runs and how the grade is produced. An edit makes a task's reference
|
||||
* runs incomparable to every other task's, and the scores still look normal,
|
||||
* so nothing downstream notices.
|
||||
*
|
||||
* A task is compared against itself as created. {@link writeManagedStamp} records
|
||||
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
|
||||
* creation, so a later mismatch is an edit made since. Tasks created before
|
||||
* stamping have no record and fall back to matching the copies this toolkit
|
||||
* ships — see {@link IntegrityStatus}.
|
||||
*
|
||||
* The toolkit appends to a task's Dockerfile itself (session staging, the
|
||||
* reference-data corpus). Those blocks are wrapped in
|
||||
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
|
||||
* comparing, so they never read as edits.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
|
||||
import { basename, join } from 'path';
|
||||
|
||||
import { banner } from './notice-banner.js';
|
||||
|
||||
/**
|
||||
* Every status is advisory. Nothing here stops a trial or a submission: an author
|
||||
* who changed one of these files did it because they didn't know we'd rather they
|
||||
* didn't, and refusing to package their work punishes a misunderstanding. The job
|
||||
* is to say so clearly, and to record it so a reviewer sees it too.
|
||||
*
|
||||
* `ok` — identical to a copy this toolkit ships, or unchanged since the
|
||||
* task was created.
|
||||
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
|
||||
* copy since. Nobody's mistake; it does mean this task's runs
|
||||
* aren't directly comparable to one built today.
|
||||
* `modified` — matches neither its baseline nor anything shipped: an edit.
|
||||
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
|
||||
* and an older release are indistinguishable.
|
||||
* `missing` — the task doesn't have the file.
|
||||
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
|
||||
* been selected yet.
|
||||
*/
|
||||
export type IntegrityStatus =
|
||||
| 'ok'
|
||||
| 'outdated'
|
||||
| 'modified'
|
||||
| 'unverifiable'
|
||||
| 'missing'
|
||||
| 'placeholder';
|
||||
|
||||
export interface FileVerdict {
|
||||
/** Task-relative path, e.g. `environment/Dockerfile`. */
|
||||
taskPath: string;
|
||||
status: IntegrityStatus;
|
||||
/** Command that restores the managed version, on `modified` / `unverifiable`. */
|
||||
restore?: string;
|
||||
}
|
||||
|
||||
export interface IntegrityReport {
|
||||
/**
|
||||
* False when this isn't a worker toolkit, or when the task was authored on
|
||||
* a different toolkit generation (see {@link isTaskFromThisToolkitGeneration})
|
||||
* — callers should skip silently.
|
||||
*/
|
||||
checked: boolean;
|
||||
files: FileVerdict[];
|
||||
/** Looks like an edit: matches neither a baseline nor anything shipped. */
|
||||
modified: FileVerdict[];
|
||||
/** Can't be told apart from an older release. */
|
||||
unverifiable: FileVerdict[];
|
||||
/** Unchanged, but a newer copy has shipped since. */
|
||||
outdated: FileVerdict[];
|
||||
}
|
||||
|
||||
interface ManagedFile {
|
||||
taskPath: string;
|
||||
/** Matches the candidate pristine filenames under `task-shared/`. */
|
||||
baselinePattern: RegExp;
|
||||
}
|
||||
|
||||
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
|
||||
const MANAGED_FILES: ManagedFile[] = [
|
||||
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
|
||||
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
|
||||
{
|
||||
taskPath: 'tests/grader-system-prompt-consolidated.md',
|
||||
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
|
||||
},
|
||||
];
|
||||
|
||||
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
|
||||
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
|
||||
|
||||
/**
|
||||
* Line shapes from toolkit releases that predate the sentinels. Deliberately
|
||||
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
|
||||
* lines" rule an edit could hide behind.
|
||||
*/
|
||||
const LEGACY_MANAGED_LINES: RegExp[] = [
|
||||
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
|
||||
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
|
||||
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
|
||||
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
|
||||
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
|
||||
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
|
||||
];
|
||||
|
||||
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
|
||||
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
|
||||
|
||||
/** Per-task stamp of the managed files as created. Lives in the task directory. */
|
||||
export const STAMP_FILENAME = '.toolkit-managed.json';
|
||||
|
||||
interface ManagedStamp {
|
||||
version: number;
|
||||
stampedAt: string;
|
||||
/** taskPath → sha256 of the stripped content. */
|
||||
files: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove toolkit-appended content so only author-authored differences remain.
|
||||
* Trailing blank lines go too — an editor adding or trimming a final newline is
|
||||
* not something to fail a trial over.
|
||||
*/
|
||||
export function stripManagedBlocks(content: string): string {
|
||||
const out: string[] = [];
|
||||
let inBlock = false;
|
||||
|
||||
// Normalize CRLF before anything else: a Windows editor or a checkout with
|
||||
// core.autocrlf rewrites every line ending, and that must not read as an edit.
|
||||
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
|
||||
if (!inBlock && SENTINEL_OPEN.test(line)) {
|
||||
inBlock = true;
|
||||
continue;
|
||||
}
|
||||
if (inBlock) {
|
||||
if (SENTINEL_CLOSE.test(line)) inBlock = false;
|
||||
continue;
|
||||
}
|
||||
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
|
||||
out.push(line);
|
||||
}
|
||||
|
||||
return out.join('\n').replace(/\s+$/, '');
|
||||
}
|
||||
|
||||
export function sha256(content: string): string {
|
||||
return createHash('sha256').update(content).digest('hex');
|
||||
}
|
||||
|
||||
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
|
||||
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
|
||||
if (!existsSync(sharedDir)) return [];
|
||||
return readdirSync(sharedDir)
|
||||
.filter((f) => pattern.test(f))
|
||||
.sort();
|
||||
}
|
||||
|
||||
/**
|
||||
* Record the managed files, so later edits are detectable. Call at task creation
|
||||
* and after a managed file is first put in place.
|
||||
*
|
||||
* A file earns a baseline only by matching a copy this toolkit ships, and an
|
||||
* entry already recorded is never rewritten. Together those mean a stamp can
|
||||
* only ever describe a pristine file: re-running this can't turn an author's
|
||||
* edit into the new baseline, and a file dropped in later (the polyglot
|
||||
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
|
||||
* baseline once it's in place.
|
||||
*
|
||||
* Returns true if anything was recorded.
|
||||
*/
|
||||
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
|
||||
const sharedDir = join(toolkitRoot, 'task-shared');
|
||||
const existing = readStamp(taskDir);
|
||||
const files: Record<string, string> = { ...(existing?.files ?? {}) };
|
||||
let added = false;
|
||||
|
||||
for (const managed of MANAGED_FILES) {
|
||||
if (files[managed.taskPath]) continue;
|
||||
const p = join(taskDir, managed.taskPath);
|
||||
if (!existsSync(p)) continue;
|
||||
const raw = readFileSync(p, 'utf-8');
|
||||
// Not a baseline: the author still has to drop in their member's base image.
|
||||
if (raw.includes(PLACEHOLDER_MARKER)) continue;
|
||||
const stripped = stripManagedBlocks(raw);
|
||||
if (!matchesShipped(sharedDir, managed, stripped)) continue;
|
||||
files[managed.taskPath] = sha256(stripped);
|
||||
added = true;
|
||||
}
|
||||
|
||||
if (!added) return false;
|
||||
|
||||
const stamp: ManagedStamp = {
|
||||
version: 1,
|
||||
stampedAt: new Date().toISOString(),
|
||||
files,
|
||||
};
|
||||
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
|
||||
return true;
|
||||
}
|
||||
|
||||
function readStamp(taskDir: string): ManagedStamp | null {
|
||||
const stampPath = join(taskDir, STAMP_FILENAME);
|
||||
if (!existsSync(stampPath)) return null;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
|
||||
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
|
||||
} catch {
|
||||
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* How to restore a managed file, or undefined when this toolkit ships no copy to
|
||||
* restore from. Only the Dockerfile can have several candidates (one per member).
|
||||
*/
|
||||
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
|
||||
const dest = `harbor-tasks/${slug}/${taskPath}`;
|
||||
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
|
||||
if (candidates.length > 1) {
|
||||
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
|
||||
}
|
||||
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
|
||||
// author to copy a Dockerfile over their grader prompt.
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Render a restore line, saying so plainly when there is nothing to restore from. */
|
||||
function restoreLine(f: FileVerdict): string {
|
||||
return f.restore
|
||||
? ` ${f.restore}`
|
||||
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
|
||||
}
|
||||
|
||||
/** Does this content match a pristine copy the toolkit ships? */
|
||||
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
|
||||
return baselineCandidates(sharedDir, managed.baselinePattern).some(
|
||||
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Was this task created by this toolkit generation? Task creation (the packed
|
||||
* scaffold's task.toml and snapshot-to-task) writes `[metadata].toolkit_version`;
|
||||
* a task directory without the key was authored on a different toolkit
|
||||
* generation and grades with the assets frozen in its own tests/ directory, so
|
||||
* comparing those against this toolkit's copies would report drift that is not
|
||||
* an edit. Presence-based on purpose: wall-clock stamps cannot separate the
|
||||
* generations, because tasks from an earlier generation are completed after
|
||||
* later kits ship.
|
||||
*
|
||||
* A regex rather than a TOML parser, for the same shipped-dependency reason as
|
||||
* input-checksums.ts readGitref: the one `toolkit_version` key in a task.toml
|
||||
* is `[metadata].toolkit_version`.
|
||||
*/
|
||||
export function isTaskFromThisToolkitGeneration(taskDir: string): boolean {
|
||||
const tomlPath = join(taskDir, 'task.toml');
|
||||
if (!existsSync(tomlPath)) return false;
|
||||
try {
|
||||
return /^\s*toolkit_version\s*=/m.test(readFileSync(tomlPath, 'utf-8'));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare a task's managed files against its creation-time stamp.
|
||||
*
|
||||
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
|
||||
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
|
||||
*/
|
||||
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
|
||||
const sharedDir = join(toolkitRoot, 'task-shared');
|
||||
|
||||
// Without task-shared/ there is nothing to compare against; report "not
|
||||
// checked" so callers no-op rather than reporting three phantom failures.
|
||||
if (!existsSync(sharedDir)) {
|
||||
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
|
||||
}
|
||||
|
||||
// A task authored on a different toolkit generation grades with the assets
|
||||
// frozen in its own tests/ directory. Comparing those against this toolkit's
|
||||
// copies would report drift that is not an edit — and the printed remedy
|
||||
// (restore the current copy) would change how that task grades. Skip it.
|
||||
if (!isTaskFromThisToolkitGeneration(taskDir)) {
|
||||
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
|
||||
}
|
||||
|
||||
const slug = basename(taskDir);
|
||||
const stamp = readStamp(taskDir);
|
||||
const files: FileVerdict[] = [];
|
||||
|
||||
for (const managed of MANAGED_FILES) {
|
||||
const taskFile = join(taskDir, managed.taskPath);
|
||||
if (!existsSync(taskFile)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'missing' });
|
||||
continue;
|
||||
}
|
||||
|
||||
const raw = readFileSync(taskFile, 'utf-8');
|
||||
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
|
||||
const restore = restoreCommand(managed.taskPath, candidates, slug);
|
||||
const stripped = stripManagedBlocks(raw);
|
||||
|
||||
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
|
||||
// cannot be an author edit, whatever the stamp says — and asking the stamp first
|
||||
// is what used to make restoring the current copy (which is exactly what we tell
|
||||
// authors to do) look like an edit, with no way out.
|
||||
if (matchesShipped(sharedDir, managed, stripped)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'ok' });
|
||||
continue;
|
||||
}
|
||||
|
||||
const expected = stamp?.files[managed.taskPath];
|
||||
if (expected) {
|
||||
// Matches its baseline but nothing shipped: untouched by the author, and the
|
||||
// toolkit has moved on since. Worth saying, nobody's fault.
|
||||
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
|
||||
files.push({ taskPath: managed.taskPath, status, restore });
|
||||
continue;
|
||||
}
|
||||
|
||||
// Checked after the stamp so that adding this marker to a file that HAS a
|
||||
// baseline can't exempt it from the comparison.
|
||||
if (raw.includes(PLACEHOLDER_MARKER)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
|
||||
continue;
|
||||
}
|
||||
|
||||
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
|
||||
}
|
||||
|
||||
return {
|
||||
checked: true,
|
||||
files,
|
||||
modified: files.filter((f) => f.status === 'modified'),
|
||||
unverifiable: files.filter((f) => f.status === 'unverifiable'),
|
||||
outdated: files.filter((f) => f.status === 'outdated'),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Human-readable report. Returns '' when there is nothing worth saying, so callers
|
||||
* can `if (msg) print(msg)`.
|
||||
*
|
||||
* Deliberately not phrased as a refusal. An author who changed one of these files
|
||||
* almost always did it to get unstuck, not knowing we'd rather they told us — so
|
||||
* this explains what it means for their task and what restoring would do, and then
|
||||
* lets them get on with it.
|
||||
*/
|
||||
export function formatIntegrityReport(report: IntegrityReport): string {
|
||||
const sections: string[] = [];
|
||||
|
||||
if (report.modified.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These files look edited since this task was created, and the toolkit manages',
|
||||
'them — they set up how the trial runs and how the grade is produced, so they',
|
||||
"have to be identical across every task. Yours aren't, which makes this task's",
|
||||
"runs hard to compare with everyone else's:",
|
||||
'',
|
||||
...report.modified.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
'Restoring the shipped version puts that right:',
|
||||
...report.modified.map(restoreLine),
|
||||
'',
|
||||
'If you changed one to work around a problem — a missing package, a grader that',
|
||||
"wouldn't run — please tell us about the problem instead. It almost certainly",
|
||||
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
|
||||
'Nothing here stops you running trials or submitting.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.outdated.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These files are unchanged, but the toolkit has shipped newer copies since this',
|
||||
'task was created:',
|
||||
'',
|
||||
...report.outdated.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
"You haven't done anything wrong. It does mean this task was run and graded with",
|
||||
"older versions than a task built today, so its scores aren't directly",
|
||||
'comparable. To line them up, restore the current copies and re-run your trials:',
|
||||
...report.outdated.map(restoreLine),
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.unverifiable.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
"These files don't match the copies this toolkit ships, and this task has no",
|
||||
'record of what they looked like when it was created:',
|
||||
'',
|
||||
...report.unverifiable.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
'Two things look like this and we cannot tell them apart: a task created on an',
|
||||
'earlier toolkit release (nothing to fix, though its scores are not directly',
|
||||
'comparable to a task built today), or a file that was edited. Either way,',
|
||||
'restoring the current copy and re-running your trials is what makes this task',
|
||||
"comparable to everyone else's:",
|
||||
...report.unverifiable.map(restoreLine),
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
return sections.join('\n\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap a report in a banner loud enough to survive a scrollback.
|
||||
*
|
||||
* Nothing blocks any more, so this notice is the entire mechanism — and an
|
||||
* unframed paragraph among build output is one a reasonable person scrolls past.
|
||||
*/
|
||||
export function bannerize(message: string, report: IntegrityReport): string {
|
||||
return banner(
|
||||
message,
|
||||
report.modified.length > 0
|
||||
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
|
||||
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!'
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,217 @@
|
||||
/**
|
||||
* toolkit-script-integrity.ts — detect edits to the toolkit's own scripts.
|
||||
*
|
||||
* Sibling of task-infra-integrity.ts, which covers a task's managed files. This
|
||||
* covers `scripts/`. The scripts never ship with a task, so an edit can't reach
|
||||
* the delivered workspace — but their OUTPUT does: `build-workspace.sh` alone
|
||||
* stages `tests/test-commands.sh` (the deterministic checks behind the
|
||||
* correctness score), writes the Dockerfile's toolkit-managed blocks, and
|
||||
* records the managed stamp and input checksums. Nothing downstream re-derives
|
||||
* those, and the reference runs can't be re-derived at all.
|
||||
*
|
||||
* The baseline is a manifest written at package time ({@link writeScriptManifest}),
|
||||
* so it ships in the same zip as the scripts it describes. That removes the
|
||||
* ambiguity a task's managed files have: there is no "created on an older
|
||||
* release" case to tell apart, so a hash mismatch is an edit. Files absent from
|
||||
* the manifest are ignored, which keeps a worker's own helper script — or a
|
||||
* `__pycache__` left by a harbor run — from ever being reported.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
|
||||
import { join, relative } from 'path';
|
||||
|
||||
import { banner } from './notice-banner.js';
|
||||
|
||||
/** Manifest of the shipped `scripts/` tree. Lives at the toolkit root. */
|
||||
export const SCRIPT_MANIFEST_FILENAME = '.toolkit-scripts.json';
|
||||
|
||||
const MANIFEST_VERSION = 1;
|
||||
|
||||
/** Runtime droppings, never part of the shipped tree. */
|
||||
const IGNORED_DIRS = new Set(['__pycache__', 'node_modules', '.git']);
|
||||
const IGNORED_FILES = /\.(pyc|pyo)$/;
|
||||
|
||||
/**
|
||||
* `modified` — content differs from what shipped: an edit.
|
||||
* `missing` — shipped, but no longer on disk.
|
||||
* `ok` — unchanged.
|
||||
*/
|
||||
export type ScriptStatus = 'ok' | 'modified' | 'missing';
|
||||
|
||||
export interface ScriptVerdict {
|
||||
/** Toolkit-relative path, e.g. `scripts/build-workspace.sh`. */
|
||||
path: string;
|
||||
status: ScriptStatus;
|
||||
}
|
||||
|
||||
export interface ScriptIntegrityReport {
|
||||
/** False when no manifest ships — callers should skip silently. */
|
||||
checked: boolean;
|
||||
files: ScriptVerdict[];
|
||||
modified: ScriptVerdict[];
|
||||
missing: ScriptVerdict[];
|
||||
}
|
||||
|
||||
interface ScriptManifest {
|
||||
version: number;
|
||||
generatedAt: string;
|
||||
/** Toolkit-relative path → sha256 of the normalized content. */
|
||||
files: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Line endings and trailing whitespace are normalized away: a Windows editor, a
|
||||
* checkout with core.autocrlf, or a formatter trimming a final newline must not
|
||||
* read as an edit.
|
||||
*/
|
||||
function hashContent(content: string): string {
|
||||
return createHash('sha256')
|
||||
.update(content.replace(/\r\n/g, '\n').replace(/\s+$/, ''))
|
||||
.digest('hex');
|
||||
}
|
||||
|
||||
/** Every shipped file under `scripts/`, as toolkit-relative paths. */
|
||||
function walkScripts(dir: string, toolkitRoot: string): string[] {
|
||||
if (!existsSync(dir)) return [];
|
||||
const out: string[] = [];
|
||||
|
||||
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) =>
|
||||
a.name.localeCompare(b.name)
|
||||
)) {
|
||||
const abs = join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
if (!IGNORED_DIRS.has(entry.name)) out.push(...walkScripts(abs, toolkitRoot));
|
||||
continue;
|
||||
}
|
||||
if (!entry.isFile() || IGNORED_FILES.test(entry.name)) continue;
|
||||
out.push(relative(toolkitRoot, abs));
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Record the shipped `scripts/` tree. Call at package time, once the tree is
|
||||
* fully staged — anything written to `scripts/` afterwards reads as an edit.
|
||||
*
|
||||
* Returns the number of files recorded.
|
||||
*/
|
||||
export function writeScriptManifest(toolkitRoot: string): number {
|
||||
const files: Record<string, string> = {};
|
||||
|
||||
for (const rel of walkScripts(join(toolkitRoot, 'scripts'), toolkitRoot)) {
|
||||
files[rel] = hashContent(readFileSync(join(toolkitRoot, rel), 'utf-8'));
|
||||
}
|
||||
|
||||
const manifest: ScriptManifest = {
|
||||
version: MANIFEST_VERSION,
|
||||
generatedAt: new Date().toISOString(),
|
||||
files,
|
||||
};
|
||||
writeFileSync(
|
||||
join(toolkitRoot, SCRIPT_MANIFEST_FILENAME),
|
||||
`${JSON.stringify(manifest, null, 2)}\n`
|
||||
);
|
||||
return Object.keys(files).length;
|
||||
}
|
||||
|
||||
function readManifest(toolkitRoot: string): ScriptManifest | null {
|
||||
const p = join(toolkitRoot, SCRIPT_MANIFEST_FILENAME);
|
||||
if (!existsSync(p)) return null;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(p, 'utf-8')) as ScriptManifest;
|
||||
if (parsed?.version !== MANIFEST_VERSION) return null;
|
||||
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
|
||||
} catch {
|
||||
// A corrupt manifest is treated as no manifest rather than blocking a trial.
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare the toolkit's `scripts/` tree against the manifest it shipped with.
|
||||
*
|
||||
* @param toolkitRoot Absolute path to the toolkit root (holds `scripts/`).
|
||||
*/
|
||||
export function checkToolkitScriptIntegrity(toolkitRoot: string): ScriptIntegrityReport {
|
||||
const manifest = readManifest(toolkitRoot);
|
||||
if (!manifest) return { checked: false, files: [], modified: [], missing: [] };
|
||||
|
||||
const files: ScriptVerdict[] = Object.entries(manifest.files).map(([path, expected]) => {
|
||||
const abs = join(toolkitRoot, path);
|
||||
if (!existsSync(abs)) return { path, status: 'missing' as const };
|
||||
const status = hashContent(readFileSync(abs, 'utf-8')) === expected ? 'ok' : 'modified';
|
||||
return { path, status };
|
||||
});
|
||||
|
||||
return {
|
||||
checked: true,
|
||||
files,
|
||||
modified: files.filter((f) => f.status === 'modified'),
|
||||
missing: files.filter((f) => f.status === 'missing'),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Human-readable report. Returns '' when there is nothing worth saying, so callers
|
||||
* can `if (msg) print(msg)`.
|
||||
*
|
||||
* Deliberately not phrased as a refusal, for the same reason the managed-file
|
||||
* notice isn't: an author who changed one of these did it to get unstuck, and the
|
||||
* fix they needed almost certainly belongs in the toolkit rather than in their copy.
|
||||
*/
|
||||
export function formatScriptIntegrityReport(report: ScriptIntegrityReport): string {
|
||||
const sections: string[] = [];
|
||||
|
||||
if (report.modified.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These toolkit scripts look edited:',
|
||||
'',
|
||||
...report.modified.map((f) => ` ${f.path}`),
|
||||
'',
|
||||
"They aren't part of any task, so an edit is easy to miss — but what they write",
|
||||
'is. Building a task stages its deterministic checks, fills in parts of its',
|
||||
'Dockerfile, and records the checksums a reviewer reads; a script that does any of',
|
||||
'that differently produces a task that looks normal and behaves differently from',
|
||||
'every other one.',
|
||||
'',
|
||||
'Re-extracting the toolkit zip over your copy restores them. Your tasks, snapshots',
|
||||
'and reference runs are untouched by that.',
|
||||
'',
|
||||
'If you changed one to work around a problem — a build that would not run, a',
|
||||
'missing dependency — please tell us about the problem instead. It almost',
|
||||
'certainly affects other authors too, and the fix belongs in the toolkit.',
|
||||
'Nothing here stops you running trials or submitting.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.missing.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These toolkit scripts shipped with this release but are no longer here:',
|
||||
'',
|
||||
...report.missing.map((f) => ` ${f.path}`),
|
||||
'',
|
||||
'Something that depends on one will fail partway through rather than up front.',
|
||||
'Re-extract the toolkit zip over your copy to put them back.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
return sections.join('\n\n');
|
||||
}
|
||||
|
||||
/** The full notice, bannered and ready to write to stderr, or '' if all is well. */
|
||||
export function scriptIntegrityNotice(report: ScriptIntegrityReport): string {
|
||||
const message = formatScriptIntegrityReport(report);
|
||||
if (!message) return '';
|
||||
|
||||
const headline =
|
||||
report.modified.length > 0
|
||||
? '!! TOOLKIT SCRIPTS LOOK EDITED — PLEASE READ !!'
|
||||
: '!! TOOLKIT SCRIPTS ARE MISSING — PLEASE READ !!';
|
||||
return banner(message, headline);
|
||||
}
|
||||
@@ -0,0 +1,304 @@
|
||||
/**
|
||||
* Tests for tree-permissions.ts.
|
||||
*
|
||||
* The load-bearing case is the one from the field report: a directory that came
|
||||
* across without its search bit makes `tar` fail with `Cannot stat` on the files
|
||||
* *inside* it, so the repair has to fix directory modes, not just ownership.
|
||||
* These tests run unprivileged, so they exercise the mode axis for real and the
|
||||
* ownership axis only as far as an unprivileged process can (target resolution +
|
||||
* graceful EPERM), which is the same shape CI runs in. One case needs root and
|
||||
* skips otherwise; the rest hold under either uid, which is why the fixtures that
|
||||
* must look human-owned say so with `ownedByHuman` instead of relying on the
|
||||
* caller's uid.
|
||||
*/
|
||||
import assert from 'node:assert/strict';
|
||||
import {
|
||||
chmodSync,
|
||||
chownSync,
|
||||
mkdirSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
symlinkSync,
|
||||
writeFileSync,
|
||||
} from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { test } from 'node:test';
|
||||
|
||||
import {
|
||||
didRepair,
|
||||
manualRepairHint,
|
||||
normalizeTreePermissions,
|
||||
resolveWorkspaceOwner,
|
||||
} from './tree-permissions';
|
||||
|
||||
function scratch(name: string): string {
|
||||
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
mkdirSync(dir, { recursive: true });
|
||||
return dir;
|
||||
}
|
||||
|
||||
const RUNNING_AS_ROOT = process.getuid?.() === 0;
|
||||
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
|
||||
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
|
||||
|
||||
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
|
||||
function ownedByHuman(path: string): string {
|
||||
chownSync(path, HUMAN_UID, HUMAN_GID);
|
||||
return path;
|
||||
}
|
||||
|
||||
test('restores the search bit on a directory that lost it', () => {
|
||||
const root = scratch('searchbit');
|
||||
const models = join(root, 'agent-output', 'app', 'models');
|
||||
mkdirSync(models, { recursive: true });
|
||||
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
|
||||
// r-- : readdir works, so tar can NAME the file, but stat is refused.
|
||||
chmodSync(models, 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
|
||||
assert.ok(report.modeFixed.some((p) => p === models));
|
||||
assert.ok(didRepair(report));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('recurses into a directory it had to widen first', () => {
|
||||
const root = scratch('recurse');
|
||||
const inner = join(root, 'locked', 'deeper');
|
||||
mkdirSync(inner, { recursive: true });
|
||||
const leaf = join(inner, 'leaf.rb');
|
||||
writeFileSync(leaf, 'x\n');
|
||||
chmodSync(leaf, 0o000);
|
||||
chmodSync(inner, 0o400);
|
||||
chmodSync(join(root, 'locked'), 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
// Only reachable if the walk widened each parent before descending.
|
||||
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
|
||||
assert.ok(report.modeFixed.includes(leaf));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('leaves already-correct trees untouched', () => {
|
||||
const root = scratch('noop');
|
||||
mkdirSync(join(root, 'sub'), { recursive: true });
|
||||
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.deepEqual(report.modeFixed, [], 'no mode changes');
|
||||
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
|
||||
assert.deepEqual(report.failures, []);
|
||||
assert.equal(didRepair(report), false);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('does not widen group/other beyond what was already there', () => {
|
||||
const root = ownedByHuman(scratch('narrow'));
|
||||
const f = join(root, 'secret.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o000);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: root });
|
||||
|
||||
const mode = statSync(f).mode & 0o777;
|
||||
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('ignores symlinks rather than following them out of the tree', () => {
|
||||
const root = scratch('symlink');
|
||||
const outside = scratch('symlink-outside');
|
||||
const victim = join(outside, 'victim.txt');
|
||||
writeFileSync(victim, 'x\n');
|
||||
chmodSync(victim, 0o000);
|
||||
symlinkSync(outside, join(root, 'link'));
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
|
||||
assert.deepEqual(report.failures, []);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
rmSync(outside, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('never throws on a missing root, and reports it', () => {
|
||||
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
|
||||
assert.equal(report.failures.length, 1);
|
||||
assert.equal(report.failures[0].reason, 'ENOENT');
|
||||
});
|
||||
|
||||
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
|
||||
const root = scratch('owner');
|
||||
const owner = resolveWorkspaceOwner(root);
|
||||
assert.ok(owner, 'resolved');
|
||||
const st = statSync(root);
|
||||
assert.equal(owner.uid, st.uid);
|
||||
assert.equal(owner.gid, st.gid);
|
||||
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('never chowns TO root, even when the owner ref is root-owned', () => {
|
||||
// The regression this guards: workspace root owned by root (unzipped with
|
||||
// sudo) while the task files are correctly owned by the human. Chowning to the
|
||||
// ref's owner would inflict the very lockout this module prevents. `/` is
|
||||
// root-owned on every platform we run on, so it's a stable stand-in.
|
||||
const root = scratch('root-ref');
|
||||
const f = join(root, 'mine.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
const beforeUid = statSync(f).uid;
|
||||
|
||||
const report = normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(report.target?.uid, 0, 'resolved a root target');
|
||||
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
|
||||
assert.deepEqual(report.failures, [], 'and did not fail trying');
|
||||
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('still normalizes modes when the chown target is root', () => {
|
||||
const root = scratch('root-ref-modes');
|
||||
const sub = join(root, 'sub');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
writeFileSync(join(sub, 'f.txt'), 'x\n');
|
||||
chmodSync(sub, 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
|
||||
assert.ok(report.modeFixed.includes(sub));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
|
||||
// The complement of the case below: we declined to chown, but these entries are
|
||||
// already the human's, so owner bits reach them and nothing should be widened.
|
||||
const root = ownedByHuman(scratch('root-ref-narrow'));
|
||||
const f = join(root, 'mine.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o600);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test(
|
||||
'grants read+search to group and other on files stranded root-owned',
|
||||
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
|
||||
() => {
|
||||
// The worker authoring container: root process, root-owned workspace. The chown
|
||||
// is declined, so owner bits land on root and the human — a different uid in
|
||||
// Explore and on a WSL host — is still locked out of a --w------- capture.
|
||||
const root = scratch('stranded');
|
||||
const sub = join(root, 'agent-output');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
const f = join(sub, 'answer.md');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o200);
|
||||
chmodSync(sub, 0o300);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(
|
||||
statSync(f).mode & 0o777,
|
||||
0o644,
|
||||
'file readable by everyone, writable by none but root'
|
||||
);
|
||||
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
|
||||
}
|
||||
);
|
||||
|
||||
test('walks a tree as deep as the filesystem allows', () => {
|
||||
const root = scratch('deep');
|
||||
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
|
||||
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
|
||||
// call-stack limit. So this isn't a stack test; it just pins that a deep,
|
||||
// narrow tree walks cleanly end to end.
|
||||
let path = root;
|
||||
for (let i = 0; i < 250; i++) {
|
||||
path = join(path, `d${i}`);
|
||||
}
|
||||
mkdirSync(path, { recursive: true });
|
||||
writeFileSync(join(path, 'leaf.txt'), 'x\n');
|
||||
chmodSync(join(path, 'leaf.txt'), 0o000);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
|
||||
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('a failure in one subtree does not abandon the rest', () => {
|
||||
const root = scratch('partial');
|
||||
const good = join(root, 'good');
|
||||
mkdirSync(good, { recursive: true });
|
||||
const goodFile = join(good, 'f.txt');
|
||||
writeFileSync(goodFile, 'x\n');
|
||||
chmodSync(goodFile, 0o000);
|
||||
// A dangling symlink and a vanished path both produce per-entry trouble.
|
||||
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
|
||||
assert.ok(report.modeFixed.includes(goodFile));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('reports rather than throws when the root is a file, not a directory', () => {
|
||||
const root = scratch('file-root');
|
||||
const f = join(root, 'lonely.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o000);
|
||||
|
||||
const report = normalizeTreePermissions(f);
|
||||
|
||||
assert.equal(statSync(f).mode & 0o600, 0o600);
|
||||
assert.deepEqual(report.failures, []);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
|
||||
const root = scratch('killswitch');
|
||||
const sub = join(root, 'sub');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
const f = join(sub, 'f.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o000);
|
||||
chmodSync(sub, 0o400);
|
||||
|
||||
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
|
||||
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
|
||||
try {
|
||||
const report = normalizeTreePermissions(root);
|
||||
assert.equal(report.skipped, true);
|
||||
assert.deepEqual(report.modeFixed, []);
|
||||
assert.deepEqual(report.ownerFixed, []);
|
||||
assert.deepEqual(report.failures, []);
|
||||
assert.equal(didRepair(report), false);
|
||||
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
|
||||
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
|
||||
}
|
||||
chmodSync(sub, 0o700);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('manual hint repairs both axes, ownership first', () => {
|
||||
const hint = manualRepairHint('harbor-tasks/my-slug');
|
||||
assert.match(hint, /chown -R/);
|
||||
assert.match(hint, /chmod -R u\+rwX/);
|
||||
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
|
||||
});
|
||||
@@ -0,0 +1,126 @@
|
||||
/**
|
||||
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
|
||||
*
|
||||
* Files captured from a task run can arrive owned by another user, or with a
|
||||
* directory missing the permission needed to walk into it. Packaging then fails
|
||||
* with `Cannot stat: Permission denied`. This repairs both.
|
||||
*
|
||||
* Grants owner rwX only, never group or other. Never throws, and never hands
|
||||
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
|
||||
*/
|
||||
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
export interface NormalizeReport {
|
||||
/** Paths whose owner was changed. */
|
||||
ownerFixed: string[];
|
||||
/** Paths whose mode gained owner rwX. */
|
||||
modeFixed: string[];
|
||||
/** Paths we wanted to change but could not, with the errno. */
|
||||
failures: { path: string; reason: string }[];
|
||||
/** Resolved target owner, or null if it couldn't be determined. */
|
||||
target: { uid: number; gid: number } | null;
|
||||
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
|
||||
skipped?: boolean;
|
||||
}
|
||||
|
||||
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
|
||||
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
|
||||
try {
|
||||
const st = statSync(ownerRef);
|
||||
return { uid: st.uid, gid: st.gid };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
|
||||
*
|
||||
* `stranded` means the file stays root-owned because we have no non-root owner to
|
||||
* give it to. Owner bits then help nobody — whoever has to read it is a different
|
||||
* user — so read and search are granted more widely. Never write, never +x on files.
|
||||
*/
|
||||
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
|
||||
const owner = isDir ? 0o700 : 0o600;
|
||||
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Give every entry under `root` to the workspace owner and make sure that owner
|
||||
* can read and traverse it. Symlinks are skipped. Repairs what it can and
|
||||
* reports what it couldn't; it never throws and never blocks its caller.
|
||||
*/
|
||||
export function normalizeTreePermissions(
|
||||
root: string,
|
||||
options: { ownerRef?: string } = {}
|
||||
): NormalizeReport {
|
||||
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
|
||||
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
|
||||
}
|
||||
|
||||
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
|
||||
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
|
||||
|
||||
// Never hand files to root — that would lock the owner out rather than help.
|
||||
const chownTarget = target && target.uid !== 0 ? target : null;
|
||||
|
||||
try {
|
||||
const stack: string[] = [root];
|
||||
while (stack.length > 0) {
|
||||
const path = stack.pop() as string;
|
||||
|
||||
let st;
|
||||
try {
|
||||
st = lstatSync(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
|
||||
continue;
|
||||
}
|
||||
if (st.isSymbolicLink()) continue;
|
||||
|
||||
const isDir = st.isDirectory();
|
||||
|
||||
// Mode first: a directory we can't search is one we can't descend into.
|
||||
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
|
||||
if (wanted !== st.mode) {
|
||||
try {
|
||||
chmodSync(path, wanted);
|
||||
report.modeFixed.push(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
|
||||
}
|
||||
}
|
||||
|
||||
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
|
||||
try {
|
||||
chownSync(path, chownTarget.uid, chownTarget.gid);
|
||||
report.ownerFixed.push(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
|
||||
}
|
||||
}
|
||||
|
||||
if (!isDir) continue;
|
||||
try {
|
||||
for (const entry of readdirSync(path)) stack.push(join(path, entry));
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
|
||||
}
|
||||
|
||||
return report;
|
||||
}
|
||||
|
||||
/** True when something was actually repaired. */
|
||||
export function didRepair(report: NormalizeReport): boolean {
|
||||
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
|
||||
}
|
||||
|
||||
/** The command to run on your host if we couldn't fix it ourselves. */
|
||||
export function manualRepairHint(path: string): string {
|
||||
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
/**
|
||||
* record-detector-inputs.ts — stamp detector report(s) with the checksums of
|
||||
* the task inputs they assessed.
|
||||
*
|
||||
* Run this right after a detector skill writes (or rewrites)
|
||||
* harbor-tasks/<slug>/detectors/<detector-name>.md. It records a sha256
|
||||
* capture of the task inputs next to the report, as
|
||||
* detectors/<detector-name>.inputs.json, so submit-task.ts can tell by
|
||||
* content — not by file timestamp — whether the report still matches the
|
||||
* task being packaged.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/record-detector-inputs.ts <task-slug> <detector-name> [detector-name...]
|
||||
* npx tsx scripts/record-detector-inputs.ts my-cool-task detector-rubric-clarity
|
||||
*/
|
||||
|
||||
import { existsSync, mkdirSync, writeFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
import pino from 'pino';
|
||||
import pinoPretty from 'pino-pretty';
|
||||
import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import { captureTaskInputs } from './lib/input-checksums';
|
||||
|
||||
const argv = yargs(hideBin(process.argv))
|
||||
.usage(
|
||||
'$0 <slug> <detectors...>',
|
||||
'Record the task-input checksums a detector report assessed',
|
||||
(y) =>
|
||||
y
|
||||
.positional('slug', { type: 'string', demandOption: true, describe: 'Task slug' })
|
||||
.positional('detectors', {
|
||||
type: 'string',
|
||||
array: true,
|
||||
demandOption: true,
|
||||
describe: 'Detector name(s), e.g. detector-rubric-clarity',
|
||||
})
|
||||
)
|
||||
.option('json', {
|
||||
type: 'boolean',
|
||||
describe: 'Output structured JSON logs',
|
||||
default: false,
|
||||
})
|
||||
.help()
|
||||
.parseSync();
|
||||
|
||||
const log = pino(
|
||||
{ name: 'record-detector-inputs', level: 'info' },
|
||||
argv.json
|
||||
? process.stdout
|
||||
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
|
||||
);
|
||||
|
||||
const slug = argv.slug as string;
|
||||
const detectors = argv.detectors as string[];
|
||||
|
||||
const taskDir = join(process.cwd(), 'harbor-tasks', slug);
|
||||
if (!existsSync(taskDir)) {
|
||||
log.fatal({ taskDir }, 'Task directory not found');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// One capture serves every report stamped in this invocation — they all
|
||||
// assessed the same on-disk revision of the task.
|
||||
const capture = captureTaskInputs(taskDir, 'stamp');
|
||||
const detectorsDir = join(taskDir, 'detectors');
|
||||
mkdirSync(detectorsDir, { recursive: true });
|
||||
|
||||
for (const name of detectors) {
|
||||
const report = join(detectorsDir, `${name}.md`);
|
||||
if (!existsSync(report)) {
|
||||
// Stamp anyway — skills sometimes stamp before the final write lands —
|
||||
// but say so, since a stamp with no report usually means a typo'd name.
|
||||
log.warn({ report: `detectors/${name}.md` }, 'No report found for this detector name');
|
||||
}
|
||||
const stampPath = join(detectorsDir, `${name}.inputs.json`);
|
||||
writeFileSync(stampPath, JSON.stringify(capture, null, 2) + '\n');
|
||||
log.info({ stamp: `detectors/${name}.inputs.json` }, 'Recorded task-input checksums');
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
"""Pure (no-harbor) helpers for inspecting a captured reference run.
|
||||
|
||||
Kept separate from ``replay_agent.py`` (which imports ``harbor``) so this logic
|
||||
can be unit-tested with the plain devcontainer Python and shipped in the worker
|
||||
toolkit alongside the replay agent.
|
||||
|
||||
The one safety-critical helper here is :func:`captured_mutating_tools`: it tells
|
||||
the replay agent whether a run with no ``agent-output/`` is a harmless advisory
|
||||
run (the agent only read + answered in chat) or a genuine capture loss (the
|
||||
agent edited files but they weren't preserved). The replay agent grades the
|
||||
former from the captured transcript and refuses the latter.
|
||||
|
||||
Structured edit tools (``Write``/``Edit``/``MultiEdit``/``NotebookEdit``) are
|
||||
obvious. ``Bash`` is the subtle one: a shell call can mutate the workspace
|
||||
(``rm``, ``mv``, ``sed -i``, ``echo … > f`` …) just as easily as it can read it.
|
||||
So a ``Bash`` call is treated as **potentially mutating unless the command is
|
||||
verifiably read-only** (:func:`bash_mutates`) — the safe direction: an unknown
|
||||
command counts as a mutation, so we never silently grade a run that lost edits.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
# Structured tools that always mutate the workspace.
|
||||
MUTATING_TOOLS = frozenset({"Write", "Edit", "MultiEdit", "NotebookEdit"})
|
||||
|
||||
# Base commands that only read (or touch non-workspace state like cwd). Anything
|
||||
# NOT here — or any file-writing redirection, or `sed -i`, or a non-read-only git
|
||||
# subcommand — is treated as potentially mutating.
|
||||
_READONLY_BASH = frozenset({
|
||||
"ls", "cat", "head", "tail", "grep", "egrep", "fgrep", "rg", "ag", "find",
|
||||
"fd", "wc", "echo", "printf", "file", "stat", "pwd", "tree", "sort", "uniq",
|
||||
"cut", "tr", "awk", "jq", "yq", "less", "more", "diff", "cmp", "basename",
|
||||
"dirname", "realpath", "readlink", "true", "false", "test", "[", "date",
|
||||
"env", "printenv", "which", "type", "command", "column", "nl", "od", "xxd",
|
||||
"hexdump", "comm", "paste", "fold", "expand", "tac", "du", "df", "seq",
|
||||
"sleep", ":", "cd", "pushd", "popd", "dirs", "whoami", "hostname", "uname",
|
||||
"id", "cksum", "md5sum", "sha1sum", "sha256sum", "strings", "wc",
|
||||
})
|
||||
# git subcommands that don't write the repo/workspace.
|
||||
_READONLY_GIT_SUB = frozenset({
|
||||
"log", "diff", "status", "show", "blame", "grep", "ls-files", "ls-tree",
|
||||
"cat-file", "rev-parse", "describe", "shortlog", "reflog", "rev-list",
|
||||
"for-each-ref", "name-rev", "symbolic-ref", "whatchanged", "var", "help",
|
||||
"show-ref", "merge-base", "cherry", "count-objects", "verify-pack",
|
||||
})
|
||||
# fd-dups (2>&1, >&2, 1>&-) and /dev/null sinks are harmless; strip them before
|
||||
# looking for a real file-writing redirection.
|
||||
_HARMLESS_REDIR = re.compile(r"[0-9&]*>>?\s*(?:&\s*[0-9-]+|/dev/null)")
|
||||
# Split a command line into segments on shell separators + substitutions.
|
||||
_SEG_SPLIT = re.compile(r"\|\||&&|[|;&\n]|\$\(|`")
|
||||
_ASSIGN = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*=")
|
||||
|
||||
|
||||
def bash_mutates(command: str) -> bool:
|
||||
"""Heuristic: does this shell command potentially write the workspace?
|
||||
|
||||
Conservative by design — errs toward True (an unrecognized command, a
|
||||
file-writing redirect, `sed -i`, or a non-read-only git subcommand all count
|
||||
as mutating). Read-only exploration (ls/cat/grep/find/wc/… piped together,
|
||||
with `2>/dev/null` / `>/dev/null`) returns False."""
|
||||
if not command or not command.strip():
|
||||
return False
|
||||
# Drop fd-dups (2>&1) + /dev/null sinks up front, so they neither look like a
|
||||
# file write nor split a segment on their `&` (2>&1 → bogus "1" command).
|
||||
stripped = _HARMLESS_REDIR.sub(" ", command)
|
||||
# 1. Any remaining redirection now writes a real file.
|
||||
if ">" in stripped:
|
||||
return True
|
||||
# 2. The leading command of every segment must be read-only.
|
||||
for seg in _SEG_SPLIT.split(stripped):
|
||||
toks = seg.split()
|
||||
idx = 0
|
||||
while idx < len(toks) and _ASSIGN.match(toks[idx]): # skip VAR=val prefixes
|
||||
idx += 1
|
||||
if idx >= len(toks):
|
||||
continue
|
||||
cmd = toks[idx].rsplit("/", 1)[-1]
|
||||
rest = toks[idx + 1:]
|
||||
if cmd == "sed" and any(t == "-i" or t.startswith("-i") for t in rest):
|
||||
return True
|
||||
if cmd == "git":
|
||||
sub = next((t for t in rest if not t.startswith("-")), "")
|
||||
if sub and sub not in _READONLY_GIT_SUB:
|
||||
return True
|
||||
continue
|
||||
if cmd and cmd not in _READONLY_BASH:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _bash_command(call_args) -> str:
|
||||
if isinstance(call_args, dict):
|
||||
return str(call_args.get("command", "") or "")
|
||||
return ""
|
||||
|
||||
|
||||
def _scan_trajectory(trajectory_path: Path) -> set[str]:
|
||||
"""Mutating tool names in an ATIF agent/trajectory.json
|
||||
(steps[].tool_calls[].function_name; Bash inspected by command)."""
|
||||
found: set[str] = set()
|
||||
if not trajectory_path.exists():
|
||||
return found
|
||||
try:
|
||||
data = json.loads(trajectory_path.read_text())
|
||||
except (json.JSONDecodeError, OSError):
|
||||
return found
|
||||
for step in data.get("steps", []):
|
||||
for call in step.get("tool_calls") or []:
|
||||
name = call.get("function_name")
|
||||
if name in MUTATING_TOOLS:
|
||||
found.add(name)
|
||||
elif name == "Bash" and bash_mutates(_bash_command(call.get("arguments"))):
|
||||
found.add("Bash")
|
||||
return found
|
||||
|
||||
|
||||
def _scan_stream_json(stream_path: Path) -> set[str]:
|
||||
"""Mutating tool names in a raw stream-json claude-code.txt
|
||||
(one JSON object per line, message.content[].tool_use; Bash by input)."""
|
||||
found: set[str] = set()
|
||||
if not stream_path.exists():
|
||||
return found
|
||||
try:
|
||||
lines = stream_path.read_text(errors="ignore").splitlines()
|
||||
except OSError:
|
||||
return found
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
obj = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
message = obj.get("message") if isinstance(obj, dict) else None
|
||||
content = message.get("content") if isinstance(message, dict) else None
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
for block in content:
|
||||
if not (isinstance(block, dict) and block.get("type") == "tool_use"):
|
||||
continue
|
||||
name = block.get("name")
|
||||
if name in MUTATING_TOOLS:
|
||||
found.add(name)
|
||||
elif name == "Bash" and bash_mutates(_bash_command(block.get("input"))):
|
||||
found.add("Bash")
|
||||
return found
|
||||
|
||||
|
||||
def captured_mutating_tools(reference_run_dir: Path | str) -> set[str]:
|
||||
"""Return the file-mutating tool names found in a captured run's transcript.
|
||||
|
||||
Checks the ATIF ``agent/trajectory.json`` first, then falls back to the raw
|
||||
stream-json ``agent/claude-code.txt``, so a lossy/partial trajectory can't
|
||||
hide a real edit. ``Bash`` is included only when its command isn't verifiably
|
||||
read-only (see :func:`bash_mutates`). An empty result means the agent made no
|
||||
workspace edits — i.e. a missing ``agent-output/`` is an advisory no-op.
|
||||
"""
|
||||
ref = Path(reference_run_dir)
|
||||
return _scan_trajectory(ref / "agent" / "trajectory.json") | _scan_stream_json(
|
||||
ref / "agent" / "claude-code.txt"
|
||||
)
|
||||
37
worker-toolkit-potion-polyglot-orig/scripts/refresh-harness-auth
Executable file
37
worker-toolkit-potion-polyglot-orig/scripts/refresh-harness-auth
Executable file
@@ -0,0 +1,37 @@
|
||||
#!/bin/bash
|
||||
# Rewrite the auth FILES harnesses read their key from — and the base URL beside them —
|
||||
# off the live .env, then exec "$@".
|
||||
#
|
||||
# codex reads its key from ${CODEX_HOME:-$HOME/.codex}/auth.json, which container-create
|
||||
# wrote once from the .env of that moment — so a key rotated afterwards never reached it
|
||||
# and needed a rebuild. claude needs none of this: it has an apiKeyHelper that re-reads
|
||||
# .env per request. Interactive launches route through here so each one re-derives first.
|
||||
#
|
||||
# The base URL never rotates, so the case that matters is the one where container-create
|
||||
# could not derive it at all (no .env yet) and wrote no config: the key then refreshes
|
||||
# fine while codex still has no proxy URL and talks to the provider directly.
|
||||
#
|
||||
# Trials are unaffected either way: harbor-run re-derives OPENAI_API_KEY per invocation
|
||||
# and harbor's codex agent authenticates the sandbox from that env var, not from this file.
|
||||
set -uo pipefail
|
||||
|
||||
_scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
|
||||
|
||||
# Subshell, and every failure swallowed: a refresh that cannot run must never stop the
|
||||
# agent from starting. The auth file already on disk is the PREVIOUS key, not nothing, so
|
||||
# failing open leaves the worker exactly where they were before this wrapper existed.
|
||||
(
|
||||
set -a
|
||||
# shellcheck disable=SC1090
|
||||
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
|
||||
set +a
|
||||
# shellcheck disable=SC1091
|
||||
HARNESS_SCRIPTS_DIR="$_scripts_dir" . "$_scripts_dir/lib/harness-credentials.sh" || exit 0
|
||||
harness_setup_credentials
|
||||
harness_write_auth
|
||||
harness_refresh_config_keys
|
||||
) >/dev/null 2>&1 || true
|
||||
|
||||
# No args is a valid call: refresh only, for a lifecycle hook.
|
||||
[ "$#" -gt 0 ] || exit 0
|
||||
exec "$@"
|
||||
201
worker-toolkit-potion-polyglot-orig/scripts/replay_agent.py
Normal file
201
worker-toolkit-potion-polyglot-orig/scripts/replay_agent.py
Normal file
@@ -0,0 +1,201 @@
|
||||
"""
|
||||
Harbor agent adapter that re-grades an existing reference run by replaying
|
||||
its captured workspace state — no model calls, no agent work.
|
||||
|
||||
Reads a `reference_run_dir` pointing at a `reference-runs/<id>/` directory
|
||||
captured during a prior real trial. Inside that dir:
|
||||
|
||||
agent/trajectory.json — the grader reads this via the symlink
|
||||
/tmp/outputs/task_transcript.txt → /logs/agent/trajectory.json
|
||||
agent-output/ — files the agent created or modified (captured by
|
||||
tests/test.sh after the agent ran)
|
||||
agent-output/_HARBOR_DELETIONS.txt
|
||||
— list of tracked files the agent deleted (one path
|
||||
per line). Empty/absent when nothing was deleted.
|
||||
|
||||
On run() (after the trial's docker env is up and /workspace has the base
|
||||
state from the Dockerfile's COPY workspace/), the adapter:
|
||||
|
||||
1. Uploads agent-output/ into /workspace — overlays the agent's surviving
|
||||
edits on top of the base workspace.
|
||||
2. Applies _HARBOR_DELETIONS.txt by `rm -f`'ing each listed path under
|
||||
/workspace. (The marker file itself was uploaded in step 1; it gets
|
||||
removed too so the verifier's re-capture doesn't pick it up as
|
||||
untracked content.)
|
||||
3. Uploads trajectory.json to /logs/agent/trajectory.json so the grader
|
||||
sees the same transcript it would have on the original run.
|
||||
|
||||
Verifier then runs as it would for any real trial — exact same code path,
|
||||
exact same artifacts, just with the agent phase replaced by a deterministic
|
||||
file-overlay. See scripts/harbor-regrade for the host-side wrapper.
|
||||
|
||||
Older reference runs captured before the deletion-capture line shipped
|
||||
(verifier-deletion-capture PR #219) won't have _HARBOR_DELETIONS.txt; the
|
||||
deletion-replay step is a no-op in that case. Deletions made in those runs
|
||||
remain lost.
|
||||
"""
|
||||
|
||||
from pathlib import Path, PurePosixPath
|
||||
|
||||
from harbor.agents.base import BaseAgent
|
||||
from harbor.environments.base import BaseEnvironment
|
||||
from harbor.models.agent.context import AgentContext
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
|
||||
# Sibling module (shipped alongside in the worker toolkit; on PYTHONPATH via
|
||||
# harbor-regrade). Kept harbor-free so its logic stays unit-testable.
|
||||
from reference_run_capture import captured_mutating_tools
|
||||
|
||||
_WORKSPACE = PurePosixPath("/workspace")
|
||||
_DELETIONS_MARKER = "_HARBOR_DELETIONS.txt"
|
||||
|
||||
|
||||
class ReplayAgent(BaseAgent):
|
||||
"""Replays a captured reference run so the verifier can be re-graded
|
||||
without invoking the model again."""
|
||||
|
||||
SUPPORTS_WINDOWS: bool = False # paths below assume POSIX /workspace.
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
logs_dir,
|
||||
reference_run_dir: str,
|
||||
source_agent_import_path: str | None = None,
|
||||
source_model_name: str | None = None,
|
||||
**kwargs,
|
||||
):
|
||||
# `source_*` are provenance, not behaviour: harbor-regrade reads them off the
|
||||
# source run and passes them so THIS replay's result.json records which
|
||||
# harness and model produced the trajectory being graded. A replay reports
|
||||
# `replay_agent:ReplayAgent` with model_name null, and the source run is
|
||||
# usually deleted (a regrade is copied back over what it regraded), so a bare
|
||||
# reference_run_dir pointer does not survive as provenance.
|
||||
#
|
||||
# They must be accepted here rather than left in **kwargs: harbor records the
|
||||
# trial config's agent kwargs regardless of what the agent does with them, but
|
||||
# BaseAgent would reject the unknown keys and take every regrade down with it.
|
||||
self._source_agent_import_path = source_agent_import_path
|
||||
self._source_model_name = source_model_name
|
||||
super().__init__(logs_dir=logs_dir, **kwargs)
|
||||
ref = Path(reference_run_dir).expanduser().resolve()
|
||||
if not ref.is_dir():
|
||||
raise FileNotFoundError(f"reference_run_dir does not exist: {ref}")
|
||||
self._reference_run_dir = ref
|
||||
self._agent_output_dir = ref / "agent-output"
|
||||
self._trajectory_path = ref / "agent" / "trajectory.json"
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "replay"
|
||||
|
||||
def version(self) -> str:
|
||||
return "1.0.0"
|
||||
|
||||
async def setup(self, environment: BaseEnvironment) -> None:
|
||||
# No installation needed; the verifier brings everything it requires.
|
||||
return
|
||||
|
||||
async def run(
|
||||
self,
|
||||
instruction: str,
|
||||
environment: BaseEnvironment,
|
||||
context: AgentContext,
|
||||
) -> None:
|
||||
# Advisory tasks — the agent only reads and answers in chat — make NO
|
||||
# workspace edits, so a faithful capture of one has an empty (or, in
|
||||
# older pipelines, absent) agent-output/. That is not a data gap: the
|
||||
# deliverable is the agent's final message, captured in
|
||||
# agent/trajectory.json, which the grader reads via
|
||||
# /tmp/outputs/task_transcript.txt. So overlay captured edits when
|
||||
# present; otherwise grade the base workspace + transcript, exactly
|
||||
# what the original advisory grading saw.
|
||||
if not self._agent_output_dir.is_dir():
|
||||
mutating = captured_mutating_tools(self._reference_run_dir)
|
||||
if mutating:
|
||||
# The agent edited files but they weren't captured — grading the
|
||||
# base workspace would silently score the wrong state. Refuse.
|
||||
raise FileNotFoundError(
|
||||
f"reference_run_dir {self._reference_run_dir} has no "
|
||||
f"agent-output/ but its captured transcript shows "
|
||||
f"file-mutating tool calls {sorted(mutating)}. The agent's "
|
||||
f"workspace edits were lost at capture time, so this run "
|
||||
f"cannot be faithfully re-graded — re-capture it."
|
||||
)
|
||||
self.logger.warning(
|
||||
"reference_run %s has no agent-output/ and made no "
|
||||
"file-mutating tool calls — treating it as an advisory run and "
|
||||
"grading the base workspace + captured transcript.",
|
||||
self._reference_run_dir,
|
||||
)
|
||||
# Skip the overlay/deletion steps; fall through to trajectory upload.
|
||||
await self._upload_trajectory(environment)
|
||||
return
|
||||
|
||||
# 1. Overlay captured agent edits onto the base /workspace.
|
||||
await environment.upload_dir(
|
||||
source_dir=str(self._agent_output_dir),
|
||||
target_dir=str(_WORKSPACE),
|
||||
)
|
||||
|
||||
# 2. Apply captured deletions, if present. Read the marker from the
|
||||
# host so we don't have to shell into the container to parse it,
|
||||
# then issue per-path rm's plus a final cleanup of the marker
|
||||
# itself (which was uploaded in step 1).
|
||||
host_marker = self._agent_output_dir / _DELETIONS_MARKER
|
||||
if host_marker.exists():
|
||||
deletion_paths = [
|
||||
line.strip()
|
||||
for line in host_marker.read_text().splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
for raw in deletion_paths:
|
||||
self._validate_relative_path(raw)
|
||||
await environment.exec(
|
||||
command=f'rm -f -- "/workspace/{raw}"',
|
||||
user="root",
|
||||
)
|
||||
await environment.exec(
|
||||
command=f'rm -f -- "/workspace/{_DELETIONS_MARKER}"',
|
||||
user="root",
|
||||
)
|
||||
|
||||
# 3. Materialize the captured trajectory at the path the grader's
|
||||
# test.sh symlinks to /tmp/outputs/task_transcript.txt.
|
||||
await self._upload_trajectory(environment)
|
||||
|
||||
async def _upload_trajectory(self, environment: BaseEnvironment) -> None:
|
||||
"""Upload agent/trajectory.json to the path the grader's test.sh
|
||||
symlinks to /tmp/outputs/task_transcript.txt. Shared by the normal
|
||||
(overlay) path and the advisory (no agent-output) path."""
|
||||
if self._trajectory_path.exists():
|
||||
env_paths = EnvironmentPaths.for_os(environment.os)
|
||||
await environment.upload_file(
|
||||
source_path=str(self._trajectory_path),
|
||||
target_path=str(env_paths.agent_dir / "trajectory.json"),
|
||||
)
|
||||
else:
|
||||
# The grader's test.sh reads /tmp/outputs/task_transcript.txt,
|
||||
# which symlinks to trajectory.json. Without the file the symlink
|
||||
# dangles and the grader sees an empty transcript — so the regrade
|
||||
# will look like the agent did nothing. Yell via harbor's own
|
||||
# logger (self.logger is a child of harbor.utils.logger) so the
|
||||
# warning lands in trial.log, not a stray "replay-agent" logger
|
||||
# nothing's wired to.
|
||||
self.logger.warning(
|
||||
"reference_run %s has no agent/trajectory.json — the grader "
|
||||
"will see an empty transcript. Investigate whether the source "
|
||||
"run was produced by an older harbor that didn't write the "
|
||||
"ATIF file (or by snapshot_agent before the multi-JSONL fix).",
|
||||
self._reference_run_dir,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_relative_path(raw: str) -> None:
|
||||
"""Guard against absolute paths and `..` traversal in the deletions
|
||||
manifest. The marker should only list paths *under* the workspace
|
||||
root; anything else is a captured-data integrity problem worth
|
||||
failing loudly on."""
|
||||
if not raw or raw.startswith("/"):
|
||||
raise ValueError(f"refusing absolute path in {_DELETIONS_MARKER}: {raw!r}")
|
||||
if ".." in PurePosixPath(raw).parts:
|
||||
raise ValueError(f"refusing `..` traversal in {_DELETIONS_MARKER}: {raw!r}")
|
||||
459
worker-toolkit-potion-polyglot-orig/scripts/resolve_harness.py
Normal file
459
worker-toolkit-potion-polyglot-orig/scripts/resolve_harness.py
Normal file
@@ -0,0 +1,459 @@
|
||||
#!/usr/bin/env python3
|
||||
"""resolve_harness.py — turn a harness id + task dir into the flags a trial needs.
|
||||
|
||||
``scripts/harbor-run`` is bash and cannot parse the TOML registry, so it shells in
|
||||
here and evals the result::
|
||||
|
||||
RESOLVED="$(python3 scripts/resolve_harness.py --task-dir "$TASK_DIR")" || exit 1
|
||||
eval "$RESOLVED"
|
||||
|
||||
Python rather than TS on purpose: this ships in the worker toolkit, whose
|
||||
package.json has no ``zod``/``smol-toml``, and ``tomllib`` is stdlib — so the
|
||||
toolkit gains a harness-aware harbor-run with zero new dependencies. There is no TS
|
||||
loader: TS callers (submit-task) shell in here, so both the schema and the selection
|
||||
policy exist exactly once and there is nothing to drift.
|
||||
|
||||
Output is POSIX ``KEY='value'`` assignments (single-quoted, embedded quotes
|
||||
escaped) on stdout; everything human-facing goes to stderr, so the eval only ever
|
||||
sees assignments. A non-zero exit means "do not launch" — the point is to fail in a
|
||||
second rather than burn agent minutes on a trial that cannot produce a usable grade.
|
||||
|
||||
Refuses to resolve when:
|
||||
- the harness id is unknown or disabled
|
||||
- the harness writes no ATIF trajectory (the grader would have no transcript)
|
||||
- the task ships a session to resume but the harness cannot resume one. This is
|
||||
the important one: it is the only failure here that would otherwise look like
|
||||
SUCCESS, with the agent answering a prompt whose conversation it never saw.
|
||||
- the harness's credential env var is unset
|
||||
|
||||
``--check-model`` additionally asks the proxy whether the model is granted. Opt-in
|
||||
on purpose: it is a network call, and one in every run's critical path trades a fast
|
||||
local failure for a new way to hang. The credential check, which is free, always runs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shlex
|
||||
import sys
|
||||
import tomllib
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
|
||||
|
||||
from harness_registry import ( # noqa: E402
|
||||
Harness,
|
||||
HarnessRegistryError,
|
||||
load_harness_registry,
|
||||
)
|
||||
|
||||
# Harness used when nothing selects one. Keeps every existing caller on today's
|
||||
# behaviour, so adding harness selection changes no current run.
|
||||
DEFAULT_HARNESS = "claude-code"
|
||||
|
||||
MODELS_TIMEOUT_SEC = 20
|
||||
|
||||
|
||||
def warn(message: str) -> None:
|
||||
print(f"resolve-harness: {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def fail(message: str) -> "None":
|
||||
print(f"resolve-harness: ERROR: {message}", file=sys.stderr)
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
def is_multi_turn(task_dir: str | None) -> bool:
|
||||
"""A task is multi-turn when it ships a NON-EMPTY session to resume. Empty is
|
||||
the documented one-shot-snapshot fallback and must run cold, so size is the
|
||||
test, not existence."""
|
||||
if not task_dir:
|
||||
return False
|
||||
session = Path(task_dir) / "environment" / "session.jsonl"
|
||||
return session.is_file() and session.stat().st_size > 0
|
||||
|
||||
|
||||
def wants_browser(task_dir: str | None) -> bool:
|
||||
"""True when task.toml opts into a browser (`[metadata] browser = true`).
|
||||
|
||||
Read straight from the file rather than via tomllib: this must agree with
|
||||
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
|
||||
text match. If the two ever disagree the agent is told about a browser the image
|
||||
lacks, which is the one failure the disclosure is designed to make impossible.
|
||||
Accepts the quoted form for the same reason build-workspace.sh does."""
|
||||
if not task_dir:
|
||||
return False
|
||||
toml_path = Path(task_dir) / "task.toml"
|
||||
if not toml_path.is_file():
|
||||
return False
|
||||
try:
|
||||
text = toml_path.read_text(encoding="utf-8")
|
||||
except OSError:
|
||||
return False
|
||||
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
|
||||
|
||||
|
||||
def harness_from_task_toml(task_dir: str | None) -> str | None:
|
||||
"""The task's own `[agent] harness` — the authoritative record of which harness
|
||||
this task was authored against.
|
||||
|
||||
This is where the worker's choice lands: the snapshot flow stamps it from the CLI
|
||||
that produced the snapshot, and a manual author writes it themselves. Either way
|
||||
it is set at task-creation time, BEFORE any trial, so nothing here depends on a
|
||||
trial's output.
|
||||
|
||||
Parsed with tomllib rather than a grep: a regex would happily match a commented
|
||||
line or the wrong table, and picking the wrong harness is a silent
|
||||
wrong-agent-runs bug.
|
||||
|
||||
Returns None when the field is simply absent — the normal case for every task
|
||||
finalized before harness selection existed — so the caller falls through to the
|
||||
toolkit default.
|
||||
|
||||
But an UNPARSEABLE task.toml refuses outright rather than falling back. Those are
|
||||
different situations and treating them alike is how the wrong harness runs
|
||||
quietly: the most likely way to break this file is adding a second `[agent]`
|
||||
table instead of a `harness` line inside the existing one (tasks already carry
|
||||
`[agent] timeout_sec`), and TOML rejects a duplicate table. Falling back there
|
||||
would run claude against a task its author wrote for codex and grade it as if
|
||||
nothing were wrong.
|
||||
"""
|
||||
if not task_dir:
|
||||
return None
|
||||
path = Path(task_dir) / "task.toml"
|
||||
if not path.is_file():
|
||||
return None
|
||||
try:
|
||||
with open(path, "rb") as handle:
|
||||
doc = tomllib.load(handle)
|
||||
except (OSError, tomllib.TOMLDecodeError) as exc:
|
||||
fail(
|
||||
f"{path} could not be parsed ({exc}). Refusing to guess a harness — fix "
|
||||
f'the file. If you were adding a harness, put `harness = "..."` inside '
|
||||
f"the EXISTING [agent] table rather than starting a second one."
|
||||
)
|
||||
harness = (doc.get("agent") or {}).get("harness")
|
||||
return harness if isinstance(harness, str) and harness else None
|
||||
|
||||
|
||||
def normalize_model(harness: Harness, model: str) -> str:
|
||||
"""Model id on the wire, per the harness's declared shape."""
|
||||
if harness.model_id_shape == "provider:model":
|
||||
return model.replace("/", ":")
|
||||
return model
|
||||
|
||||
|
||||
def granted_models(harness: Harness) -> list[str] | None:
|
||||
"""Model ids the key is granted, or None when the check couldn't run."""
|
||||
base_url = os.environ.get(harness.base_url_env or "")
|
||||
key = os.environ.get(harness.key_env or "")
|
||||
if not base_url or not key:
|
||||
warn("--check-model skipped: base URL or key env is unset")
|
||||
return None
|
||||
request = urllib.request.Request(
|
||||
f"{base_url.rstrip('/')}/models", headers={"Authorization": f"Bearer {key}"}
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=MODELS_TIMEOUT_SEC) as response:
|
||||
body = json.loads(response.read().decode("utf-8"))
|
||||
except (urllib.error.URLError, TimeoutError, ValueError, OSError) as exc:
|
||||
warn(f"--check-model skipped: /models unreachable ({exc})")
|
||||
return None
|
||||
return [m["id"] for m in body.get("data", []) if isinstance(m.get("id"), str)]
|
||||
|
||||
|
||||
def assert_model_granted(harness: Harness, model: str) -> None:
|
||||
granted = granted_models(harness)
|
||||
if granted is None:
|
||||
return
|
||||
# The proxy LISTS ids provider-prefixed ("openai/gpt-5.6-sol") but 400s on that
|
||||
# form — requests take the bare id. Accept either spelling.
|
||||
bare = {g.split("/")[-1] for g in granted}
|
||||
if model not in granted and model not in bare:
|
||||
shown = ", ".join(granted[:12]) + (", …" if len(granted) > 12 else "")
|
||||
fail(
|
||||
f'Model "{model}" is not granted for this key. Granted ({len(granted)}): {shown}'
|
||||
)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
parser.add_argument(
|
||||
"--harness",
|
||||
help=f"harness id (default: the task's [agent] harness, else {DEFAULT_HARNESS})",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--task-dir",
|
||||
help="task directory; decides multi-turn from environment/session.jsonl",
|
||||
)
|
||||
parser.add_argument("--model", help="override the harness's default model")
|
||||
parser.add_argument(
|
||||
"--fast",
|
||||
action="store_true",
|
||||
help="run the trial agent in the harness's fast serving mode (higher token "
|
||||
"rate, faster output). Refuses on a harness that has none.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--check-model",
|
||||
action="store_true",
|
||||
help="also ask the proxy whether the model is granted (network call)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--authoring-installs",
|
||||
action="store_true",
|
||||
help="print '<id>\\t<cli>\\t<install>' for each harness a worker can author "
|
||||
"with, and exit. Consumed by scripts/setup-harnesses.sh so the worker "
|
||||
"containers install from the registry rather than from hardcoded lists that "
|
||||
"drift.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container-configs",
|
||||
action="store_true",
|
||||
help="print '<id>\\t<config_path>\\t<base64 container_config>' for each "
|
||||
"authoring harness that declares one, and exit. Base64 because the config is "
|
||||
"multi-line TOML and these query modes are line-oriented.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--surface",
|
||||
choices=("authoring", "explore"),
|
||||
default="authoring",
|
||||
help="which worker container --container-configs is for; explore additionally "
|
||||
"gets the capture hooks, whose commands only ship there.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--defaults",
|
||||
action="store_true",
|
||||
help="print '<id>\\t<default_model>\\t<effort_default>' for every harness, and "
|
||||
"exit. For recording what a task was authored against; nothing reads it back.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--explore-launchers",
|
||||
action="store_true",
|
||||
help="print '<id>\\t<cli>\\t<launch command>' for each authoring harness, and "
|
||||
"exit. The launch command has the registry's model, effort and agent_config "
|
||||
"already substituted, so Explore and a trial cannot disagree about them. "
|
||||
"Consumed by setup-harnesses.sh to write one launcher per harness.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skills-dirs",
|
||||
action="store_true",
|
||||
help="print '<id>\t<skills_dir>' for each authoring harness that discovers "
|
||||
"skills from a directory, and exit. Lets setup-harnesses.sh install the "
|
||||
"snapshot skill for harnesses that have no plugin system.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--auth-files",
|
||||
action="store_true",
|
||||
help="print '<id>\t<auth_path>\t<auth_key_env>' for each authoring harness that "
|
||||
"authenticates from a file rather than the environment, and exit.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--authoring-credentials",
|
||||
action="store_true",
|
||||
help="print '<id>\\t<key_env>\\t<base_url_env>\\t<proxy_path>' for each "
|
||||
"authoring harness, and exit. Lets the containers point every harness at the "
|
||||
"same proxy key on its own provider path.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--declared-harness",
|
||||
action="store_true",
|
||||
help="print ONLY the harness --task-dir's task.toml declares (empty if it "
|
||||
"declares none) and exit. Unlike the default mode this applies no fallback, so "
|
||||
"a caller can tell 'declared' from 'defaulted'. Exists so consumers without a "
|
||||
"TOML parser never hand-roll one: a regex would match a commented line or the "
|
||||
"wrong table, and the duplicate-[agent] shape is exactly the likely mistake.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--resolve-identity",
|
||||
action="append",
|
||||
default=None,
|
||||
metavar="AGENT",
|
||||
help="resolve agent identities (a result.json config.agent import_path or name) "
|
||||
"to harness ids and exit; repeatable. Prints one TAB-separated "
|
||||
"'<identity>\\t<harness-id>' line each, with an empty id when nothing claims it. "
|
||||
"Lets callers that cannot import the registry (the worker toolkit has no "
|
||||
"zod/smol-toml) still resolve through the one source of truth.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--list",
|
||||
action="store_true",
|
||||
help="print the selectable harnesses and exit (what task.toml's [agent] harness accepts)",
|
||||
)
|
||||
parser.add_argument("--registry", default=None, help="registry path (tests)")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
try:
|
||||
registry = (
|
||||
load_harness_registry(args.registry)
|
||||
if args.registry
|
||||
else load_harness_registry()
|
||||
)
|
||||
except HarnessRegistryError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
# --- read-only query modes: answer and exit, never emit assignments -------
|
||||
if args.authoring_installs:
|
||||
for harness in registry.authoring():
|
||||
print(f"{harness.id}\t{harness.cli or ''}\t{harness.install or ''}")
|
||||
return 0
|
||||
|
||||
if args.container_configs:
|
||||
import base64
|
||||
|
||||
for harness in registry.authoring():
|
||||
config = harness.container_config_text(surface=args.surface)
|
||||
if not (harness.config_path and config):
|
||||
continue
|
||||
blob = base64.b64encode(config.encode()).decode()
|
||||
print(f"{harness.id}\t{harness.config_path}\t{blob}")
|
||||
return 0
|
||||
|
||||
if args.defaults:
|
||||
for harness in registry.all():
|
||||
print(
|
||||
f"{harness.id}\t{harness.default_model or ''}\t"
|
||||
f"{harness.effort_default or ''}"
|
||||
)
|
||||
return 0
|
||||
|
||||
if args.explore_launchers:
|
||||
for harness in registry.authoring():
|
||||
print(
|
||||
f"{harness.id}\t{harness.cli or ''}\t"
|
||||
f"{harness.explore_launch_command() or ''}"
|
||||
)
|
||||
return 0
|
||||
|
||||
if args.skills_dirs:
|
||||
for harness in registry.authoring():
|
||||
if harness.skills_dir:
|
||||
print(f"{harness.id}\t{harness.skills_dir}")
|
||||
return 0
|
||||
|
||||
if args.auth_files:
|
||||
for harness in registry.authoring():
|
||||
if harness.auth_path and harness.auth_key_env:
|
||||
print(f"{harness.id}\t{harness.auth_path}\t{harness.auth_key_env}")
|
||||
return 0
|
||||
|
||||
if args.authoring_credentials:
|
||||
for harness in registry.authoring():
|
||||
print(
|
||||
f"{harness.id}\t{harness.key_env or ''}\t"
|
||||
f"{harness.base_url_env or ''}\t{harness.proxy_path or ''}"
|
||||
)
|
||||
return 0
|
||||
|
||||
if args.declared_harness:
|
||||
print(harness_from_task_toml(args.task_dir) or "")
|
||||
return 0
|
||||
|
||||
if args.resolve_identity:
|
||||
for identity in args.resolve_identity:
|
||||
harness = registry.by_import_path(identity)
|
||||
print(f"{identity}\t{harness.id if harness else ''}")
|
||||
return 0
|
||||
|
||||
if args.list:
|
||||
# Printed on stdout because it is the requested output here, not the
|
||||
# eval-able assignments — this mode is for a human, and never shelled into.
|
||||
for harness in registry.enabled():
|
||||
turns = (
|
||||
"multi-turn + single-turn"
|
||||
if harness.seed_native
|
||||
else "single-turn only"
|
||||
)
|
||||
model = harness.default_model or "(pass --model)"
|
||||
print(f"{harness.id:<16} {harness.label:<20} {turns:<24} {model}")
|
||||
return 0
|
||||
|
||||
# --- selection ------------------------------------------------------------
|
||||
# Precedence: an explicit --harness (a benchmark, or a deliberate override) beats the
|
||||
# task's own record, which is what its author chose. Everything else — every task
|
||||
# finalized before harness selection existed — is the default.
|
||||
requested = args.harness or harness_from_task_toml(args.task_dir) or DEFAULT_HARNESS
|
||||
try:
|
||||
harness = registry.require(requested)
|
||||
except HarnessRegistryError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
if not harness.enabled:
|
||||
fail(
|
||||
f'Harness "{harness.id}" is disabled in the registry (never verified here). '
|
||||
f"Enable it in scripts/harness-registry.toml once a trial has been run with it."
|
||||
)
|
||||
if not harness.writes_atif:
|
||||
fail(
|
||||
f'Harness "{harness.id}" writes no ATIF trajectory, so the grader would have '
|
||||
f"no transcript and its rewards would be meaningless."
|
||||
)
|
||||
|
||||
multi_turn = is_multi_turn(args.task_dir)
|
||||
if multi_turn and not harness.seed_native:
|
||||
fail(
|
||||
f'Task ships a session to resume, but harness "{harness.id}" cannot resume '
|
||||
f"one. Running anyway would look like a success while the agent answered a "
|
||||
f"prompt whose conversation it never saw."
|
||||
)
|
||||
|
||||
if harness.key_env and not os.environ.get(harness.key_env):
|
||||
fail(f'{harness.key_env} is unset — required by harness "{harness.id}".')
|
||||
|
||||
if args.fast and not harness.fast_kwarg:
|
||||
fail(
|
||||
f'Harness "{harness.id}" has no fast serving mode (no fast_kwarg in the '
|
||||
f"registry). Drop --fast or pick a harness that declares one."
|
||||
)
|
||||
|
||||
model = args.model or harness.default_model
|
||||
if not model:
|
||||
fail(
|
||||
f'Harness "{harness.id}" has no default_model in the registry; pass --model '
|
||||
f"explicitly."
|
||||
)
|
||||
if args.check_model:
|
||||
assert_model_granted(harness, model)
|
||||
|
||||
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
|
||||
# anything it would only echo back at the worker is said below instead.
|
||||
browser = wants_browser(args.task_dir)
|
||||
assignments = {
|
||||
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
|
||||
"MODEL": normalize_model(harness, model),
|
||||
"EFFORT_KWARG": harness.effort_kwarg,
|
||||
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
|
||||
"FAST_KWARG": harness.fast_kwarg if args.fast else "",
|
||||
}
|
||||
|
||||
warn(
|
||||
f"{harness.label} · model={assignments['MODEL']} · "
|
||||
f"{'multi-turn' if multi_turn else 'single-turn'} · "
|
||||
f"{'browser · ' if browser else ''}"
|
||||
f"{'fast · ' if args.fast else ''}"
|
||||
f"agent={assignments['AGENT_IMPORT_PATH']}"
|
||||
)
|
||||
if browser and not harness.agent_import_path_browser:
|
||||
# Not a failure: the image still gets Playwright and the agent is still told about
|
||||
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
|
||||
# letting someone infer from a log line that the opt-in was ignored entirely.
|
||||
warn(
|
||||
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
|
||||
f"The browser and its disclosure are unaffected."
|
||||
)
|
||||
if harness.flaky_hangs:
|
||||
warn(
|
||||
f"{harness.label} is known to hang with no client-side timeout on a small "
|
||||
f"fraction of trials. A silent, output-less trial is that, not a task defect."
|
||||
)
|
||||
for key, value in assignments.items():
|
||||
print(f"{key}={shlex.quote(value)}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,260 @@
|
||||
/**
|
||||
* Strip machine-identifying filesystem paths, and optional keywords, from a session
|
||||
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
|
||||
*/
|
||||
|
||||
export const DEFAULT_PLACEHOLDER = '~/repo';
|
||||
export const HOME_DIR_PLACEHOLDER = '~';
|
||||
export const REDACTION_PLACEHOLDER = '[redacted]';
|
||||
|
||||
export interface SanitizeOptions {
|
||||
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
|
||||
placeholder?: string;
|
||||
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
|
||||
forbiddenMarkers?: readonly RegExp[];
|
||||
/**
|
||||
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
|
||||
* there, so callers that know the root pass it here.
|
||||
*/
|
||||
cwdPrefix?: string;
|
||||
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
|
||||
cwdPrefixes?: readonly string[];
|
||||
/**
|
||||
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
|
||||
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
|
||||
*/
|
||||
scrubEmbeddedHomePaths?: boolean;
|
||||
}
|
||||
|
||||
export interface SanitizeResult {
|
||||
sanitized: string;
|
||||
prefixStripped: string | null;
|
||||
encodedPrefixStripped: string | null;
|
||||
homeDirStripped: string | null;
|
||||
encodedHomeDirStripped: string | null;
|
||||
embeddedPrefixStripped: string | null;
|
||||
embeddedHomeDirStripped: string | null;
|
||||
/** Replacement count per marker, keyed by the regex's source string. */
|
||||
markersScrubbed: Record<string, number>;
|
||||
}
|
||||
|
||||
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
|
||||
* Returns `''` when only the root `/` is common. */
|
||||
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
|
||||
const arr = Array.from(paths);
|
||||
if (arr.length === 0) return '';
|
||||
const splits = arr.map((p) => p.split('/'));
|
||||
const minLen = Math.min(...splits.map((s) => s.length));
|
||||
let lastShared = 0;
|
||||
for (let i = 0; i < minLen; i++) {
|
||||
const c = splits[0][i];
|
||||
if (splits.some((s) => s[i] !== c)) break;
|
||||
lastShared = i + 1;
|
||||
}
|
||||
// Only the leading empty piece matched → just the root, not useful.
|
||||
if (lastShared <= 1) return '';
|
||||
return splits[0].slice(0, lastShared).join('/');
|
||||
}
|
||||
|
||||
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
|
||||
* better to skip the home pass than strip what may be repo content. */
|
||||
export function extractHomeDir(cwdPrefix: string): string | null {
|
||||
if (!cwdPrefix.startsWith('/')) return null;
|
||||
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
|
||||
// drive letter and leave the account name in. A volume or drive root carries no
|
||||
// identity by itself, so those take the directory under it.
|
||||
const patterns: RegExp[] = [
|
||||
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
|
||||
/^\/mnt\/[^/]+\/Users\/[^/]+/,
|
||||
/^\/Users\/[^/]+/,
|
||||
/^\/home\/[^/]+/,
|
||||
/^\/Volumes\/[^/]+\/[^/]+/,
|
||||
/^\/mnt\/[^/]+\/[^/]+/,
|
||||
/^\/var\/root(?=\/|$)/,
|
||||
/^\/root(?=\/|$)/,
|
||||
];
|
||||
for (const re of patterns) {
|
||||
const m = cwdPrefix.match(re);
|
||||
if (m) return m[0];
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
|
||||
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
|
||||
export function collectCwds(raw: string): Set<string> {
|
||||
const out = new Set<string>();
|
||||
const add = (v: unknown) => {
|
||||
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
|
||||
};
|
||||
for (const line of raw.split('\n')) {
|
||||
if (!line.trim()) continue;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (typeof parsed !== 'object' || parsed === null) continue;
|
||||
const rec = parsed as { cwd?: unknown; payload?: unknown };
|
||||
add(rec.cwd);
|
||||
if (typeof rec.payload === 'object' && rec.payload !== null) {
|
||||
add((rec.payload as { cwd?: unknown }).cwd);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
|
||||
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
|
||||
// macOS/Windows display names can contain spaces, but only consume them while
|
||||
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
|
||||
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
|
||||
const EMBEDDED_HOME_RE = new RegExp(
|
||||
'(?:' +
|
||||
String.raw`\/home\/${COMP}` +
|
||||
'|' +
|
||||
String.raw`\/Users\/${USER_WITH_SPACES}` +
|
||||
'|' +
|
||||
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
|
||||
'|' +
|
||||
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
|
||||
String.raw`\/var\/root(?![^/])` +
|
||||
'|' +
|
||||
String.raw`\/root(?![^/])` +
|
||||
')' +
|
||||
String.raw`(?:\/${COMP})*`,
|
||||
'g'
|
||||
);
|
||||
|
||||
export function collectEmbeddedHomePaths(raw: string): Set<string> {
|
||||
const out = new Set<string>();
|
||||
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
|
||||
return out;
|
||||
}
|
||||
|
||||
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
|
||||
if (!needle) return haystack;
|
||||
return haystack.split(needle).join(replacement);
|
||||
}
|
||||
|
||||
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
|
||||
* is one component but `…/repo.` ending a sentence is not. */
|
||||
function continuesComponent(text: string, at: number): boolean {
|
||||
const ch = text[at];
|
||||
if (ch === undefined) return false;
|
||||
if (/[A-Za-z0-9_-]/.test(ch)) return true;
|
||||
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
|
||||
}
|
||||
|
||||
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
|
||||
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
|
||||
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
|
||||
if (!needle) return haystack;
|
||||
let out = '';
|
||||
let from = 0;
|
||||
for (;;) {
|
||||
const i = haystack.indexOf(needle, from);
|
||||
if (i === -1) return out + haystack.slice(from);
|
||||
const end = i + needle.length;
|
||||
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
|
||||
from = end;
|
||||
}
|
||||
}
|
||||
|
||||
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
|
||||
function stripBothForms(haystack: string, needle: string, replacement: string): string {
|
||||
const out = literalReplaceAll(haystack, needle, replacement);
|
||||
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
|
||||
}
|
||||
|
||||
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
|
||||
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
|
||||
const markers = opts.forbiddenMarkers ?? [];
|
||||
const cwds = collectCwds(raw);
|
||||
let working = raw;
|
||||
let prefixStripped: string | null = null;
|
||||
let encodedPrefixStripped: string | null = null;
|
||||
let homeDirStripped: string | null = null;
|
||||
let encodedHomeDirStripped: string | null = null;
|
||||
let embeddedPrefixStripped: string | null = null;
|
||||
let embeddedHomeDirStripped: string | null = null;
|
||||
|
||||
const requested = opts.cwdPrefixes?.length
|
||||
? [...opts.cwdPrefixes]
|
||||
: opts.cwdPrefix
|
||||
? [opts.cwdPrefix]
|
||||
: cwds.size > 0
|
||||
? [findLongestCommonPathPrefix(cwds)]
|
||||
: [];
|
||||
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
|
||||
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
|
||||
|
||||
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
|
||||
// sibling root's own prefix, leaving it unmatched when its turn came.
|
||||
for (const prefix of prefixes) {
|
||||
const encodedPrefix = prefix.replace(/\//g, '-');
|
||||
working = replacePrefixAtBoundary(working, prefix, placeholder);
|
||||
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
|
||||
prefixStripped ??= prefix;
|
||||
encodedPrefixStripped ??= encodedPrefix;
|
||||
}
|
||||
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
|
||||
const homeDirs = new Set(
|
||||
prefixes
|
||||
.map((p) => extractHomeDir(p))
|
||||
.filter((h): h is string => h !== null && !prefixes.includes(h))
|
||||
);
|
||||
for (const homeDir of homeDirs) {
|
||||
const encodedHomeDir = homeDir.replace(/\//g, '-');
|
||||
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
|
||||
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
|
||||
homeDirStripped ??= homeDir;
|
||||
encodedHomeDirStripped ??= encodedHomeDir;
|
||||
}
|
||||
|
||||
if (opts.scrubEmbeddedHomePaths) {
|
||||
const embedded = collectEmbeddedHomePaths(working);
|
||||
if (embedded.size > 0) {
|
||||
// Take each path's own shortest `/repo`-terminated prefix rather than a
|
||||
// common prefix, which mis-collapses when paths diverge above the root.
|
||||
const repoRoots = new Set<string>();
|
||||
const homeDirs = new Set<string>();
|
||||
for (const p of embedded) {
|
||||
const h = extractHomeDir(p);
|
||||
if (h) homeDirs.add(h);
|
||||
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
|
||||
if (m) repoRoots.add(m[1]);
|
||||
}
|
||||
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
|
||||
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
|
||||
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
|
||||
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
|
||||
embeddedPrefixStripped = sortedRoots[0] ?? null;
|
||||
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
|
||||
}
|
||||
}
|
||||
|
||||
const markersScrubbed: Record<string, number> = {};
|
||||
for (const re of markers) {
|
||||
let count = 0;
|
||||
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
|
||||
const global = new RegExp(re.source, flags);
|
||||
working = working.replace(global, () => {
|
||||
count++;
|
||||
return REDACTION_PLACEHOLDER;
|
||||
});
|
||||
if (count > 0) markersScrubbed[re.source] = count;
|
||||
}
|
||||
|
||||
return {
|
||||
sanitized: working,
|
||||
prefixStripped,
|
||||
encodedPrefixStripped,
|
||||
homeDirStripped,
|
||||
encodedHomeDirStripped,
|
||||
embeddedPrefixStripped,
|
||||
embeddedHomeDirStripped,
|
||||
markersScrubbed,
|
||||
};
|
||||
}
|
||||
34
worker-toolkit-potion-polyglot-orig/scripts/session-id.ts
Normal file
34
worker-toolkit-potion-polyglot-orig/scripts/session-id.ts
Normal file
@@ -0,0 +1,34 @@
|
||||
/**
|
||||
* Shared helper for finding a Claude Code session id inside a JSONL or
|
||||
* stream-json `.txt` file. Both formats embed the id under one of two keys:
|
||||
*
|
||||
* - `session_id` — stream-json output (`claude-code.txt` from `--print
|
||||
* --output-format=stream-json`).
|
||||
* - `sessionId` — Claude Code's internal session log (the resumable JSONL
|
||||
* under `~/.claude/projects/<key>/<id>.jsonl`).
|
||||
*
|
||||
* Real files only use one key, but if a future format ever emits both we
|
||||
* shouldn't have two callers picking different winners — so this helper is
|
||||
* the single source of truth.
|
||||
*/
|
||||
import { readFileSync } from 'fs';
|
||||
|
||||
export function readSessionId(filePath: string): string | null {
|
||||
const lines = readFileSync(filePath, 'utf8').split('\n');
|
||||
for (const line of lines) {
|
||||
if (!line) continue;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (typeof parsed !== 'object' || parsed === null) continue;
|
||||
const obj = parsed as Record<string, unknown>;
|
||||
const camel = typeof obj.sessionId === 'string' ? obj.sessionId : null;
|
||||
const snake = typeof obj.session_id === 'string' ? obj.session_id : null;
|
||||
const id = camel ?? snake;
|
||||
if (id) return id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
340
worker-toolkit-potion-polyglot-orig/scripts/setup-harnesses.sh
Normal file
340
worker-toolkit-potion-polyglot-orig/scripts/setup-harnesses.sh
Normal file
@@ -0,0 +1,340 @@
|
||||
#!/bin/bash
|
||||
# Install the harnesses a worker can author with, from scripts/harness-registry.toml.
|
||||
#
|
||||
# Source it, then call unpiped — it exports credentials, which a subshell would lose:
|
||||
#
|
||||
# . /workspace/scripts/setup-harnesses.sh
|
||||
# harness_setup_all
|
||||
#
|
||||
# Registry reading and credential derivation live in lib/harness-credentials.sh, sourced
|
||||
# below, because `harbor-run` needs those and nothing else here.
|
||||
#
|
||||
# No -e here — but this file is SOURCED, and shell options belong to the caller's shell:
|
||||
# both post-creates run with -e, so that is what is in force. An unguarded failure below
|
||||
# therefore aborts container creation, which is why every failure site is individually
|
||||
# guarded (`|| true`, `if !`) rather than relying on this line.
|
||||
set -uo pipefail
|
||||
|
||||
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
|
||||
if [ ! -f "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh" ]; then
|
||||
echo "harness-setup: FATAL — $_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh is" >&2
|
||||
echo "harness-setup: missing, so nothing here can read the registry. Every step below" >&2
|
||||
echo "harness-setup: would report a missing interpreter instead of this." >&2
|
||||
return 1 2>/dev/null || exit 1
|
||||
fi
|
||||
# shellcheck disable=SC1091
|
||||
. "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh"
|
||||
|
||||
# Every setup step reads the registry through _harness_query, and each call suppresses
|
||||
# stderr so one bad row can't abort the container. That means a BROKEN interpreter turns
|
||||
# the whole of setup into a silent no-op: no credentials, no CLIs, no config, no
|
||||
# launchers, and no error anywhere. Check it once, loudly, before any of that.
|
||||
harness_preflight() {
|
||||
local err py found=yes
|
||||
py=$(_raccoon_python) || { py=python3; found=no; }
|
||||
if ! err=$("$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" --list 2>&1 >/dev/null); then
|
||||
echo "harness-setup: FATAL — cannot read the harness registry, so no agent CLI" >&2
|
||||
echo "harness-setup: would be installed. Nothing below will run." >&2
|
||||
echo "harness-setup: interpreter: $(command -v "$py" || echo MISSING) ($("$py" -V 2>&1))" >&2
|
||||
if [ "$found" = no ]; then
|
||||
echo "harness-setup: no python3.11+ with tomllib found; set RACCOON_PYTHON to override" >&2
|
||||
fi
|
||||
echo "harness-setup: registry: $_HARNESS_REGISTRY_DIR/harness-registry.toml" >&2
|
||||
printf 'harness-setup: %s\n' "$err" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# claude installs into $HOME/.local/bin, which is not on PATH during post-create.
|
||||
case ":$PATH:" in
|
||||
*":$HOME/.local/bin:"*) ;;
|
||||
*) export PATH="$HOME/.local/bin:$PATH" ;;
|
||||
esac
|
||||
|
||||
# --- installs ----------------------------------------------------------------
|
||||
harness_install_clis() {
|
||||
local id cli install
|
||||
while IFS=$'\t' read -r id cli install; do
|
||||
[ -n "$install" ] || continue
|
||||
if command -v "$cli" >/dev/null 2>&1; then
|
||||
echo "harness-setup: $cli already installed — skipping" >&2
|
||||
continue
|
||||
fi
|
||||
echo "harness-setup: installing $id ($cli)" >&2
|
||||
# Reported as unavailable below rather than fatal.
|
||||
if ! bash -c "$install" >&2; then
|
||||
echo "harness-setup: WARNING $id failed to install — $cli will be unavailable" >&2
|
||||
fi
|
||||
done < <(_harness_query --authoring-installs 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Report which CLIs are usable. Non-zero when NONE are: one harness missing is survivable
|
||||
# (a worker uses the other), but zero means the container cannot author anything at all,
|
||||
# and that must stop setup rather than read as a couple of warnings.
|
||||
harness_report() {
|
||||
local id cli install ready=0 missing=0
|
||||
while IFS=$'\t' read -r id cli install; do
|
||||
[ -n "$cli" ] || continue
|
||||
if command -v "$cli" >/dev/null 2>&1; then
|
||||
echo " $cli — ready" >&2
|
||||
ready=$((ready + 1))
|
||||
else
|
||||
echo " $cli — NOT AVAILABLE (install failed; see above)" >&2
|
||||
missing=$((missing + 1))
|
||||
fi
|
||||
done < <(_harness_query --authoring-installs 2>/dev/null || true)
|
||||
|
||||
# A CLI on PATH with no key is worse than a missing one: it starts, then fails at the
|
||||
# first request with the harness's own auth error, which says nothing about setup.
|
||||
local id key_env base_url_env proxy_path
|
||||
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
|
||||
[ -n "$key_env" ] || continue
|
||||
if [ -z "${!key_env:-}" ]; then
|
||||
echo " $id — installed but NO CREDENTIALS: $key_env is unset." >&2
|
||||
echo " Derived from ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY; set both in .env." >&2
|
||||
fi
|
||||
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
|
||||
|
||||
if [ "$ready" -eq 0 ]; then
|
||||
echo "harness-setup: FATAL — no agent CLI installed ($missing attempted)." >&2
|
||||
echo "harness-setup: This container cannot author a task. Check the install" >&2
|
||||
echo "harness-setup: output above: the CLIs download over the network, so a" >&2
|
||||
echo "harness-setup: proxy, DNS or upstream change breaks every one at once." >&2
|
||||
return 1
|
||||
fi
|
||||
[ "$missing" -gt 0 ] && echo "harness-setup: $missing harness(es) unavailable; $ready usable" >&2
|
||||
return 0
|
||||
}
|
||||
|
||||
# --- Explore launchers -------------------------------------------------------
|
||||
# One `raccoon-explore-<cli>` per harness, aliased to its `cli`.
|
||||
harness_install_launchers() {
|
||||
local bin="$HOME/.local/bin"
|
||||
mkdir -p "$bin"
|
||||
# Read at launcher run time so the note stays a file, not a baked-in copy.
|
||||
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
|
||||
local browser_note_src="${note_src%.md}_browser.md"
|
||||
local read_note_src="${note_src%.md}_read.md"
|
||||
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
|
||||
|
||||
# Which harnesses keep their key in a file rather than reading $ENV per request. Those
|
||||
# launchers refresh it first: the file dates from container create, so a key rotated in
|
||||
# .env since then would otherwise reach the harness only after a rebuild.
|
||||
local file_auth_ids="" aid apath akey
|
||||
while IFS=$'\t' read -r aid apath akey; do
|
||||
[ -n "$apath" ] || continue
|
||||
file_auth_ids="${file_auth_ids:+$file_auth_ids }$aid"
|
||||
done < <(_harness_query --auth-files 2>/dev/null || true)
|
||||
|
||||
local id cli launch switchable refresh_line
|
||||
while IFS=$'\t' read -r id cli launch; do
|
||||
[ -n "$cli" ] && [ -n "$launch" ] || continue
|
||||
# `|| true` twice over (here and inside the script): the launcher runs under
|
||||
# `set -e`, and a failed refresh must not cost the worker their agent.
|
||||
if [[ " $file_auth_ids " == *" $id "* ]]; then
|
||||
refresh_line="\"$_HARNESS_REGISTRY_DIR/refresh-harness-auth\" || true"
|
||||
else
|
||||
refresh_line=""
|
||||
fi
|
||||
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
|
||||
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
|
||||
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
|
||||
# so a browser task needs nothing added and the flag has nothing to switch.
|
||||
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
|
||||
# which every launch line references, and every harness would look switchable.
|
||||
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
|
||||
switchable=1
|
||||
else
|
||||
switchable=0
|
||||
fi
|
||||
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
|
||||
#!/bin/bash
|
||||
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
|
||||
set -euo pipefail
|
||||
if [ -f "$note_src" ]; then
|
||||
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
|
||||
else
|
||||
RACCOON_TOOLSET_NOTE=""
|
||||
fi
|
||||
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
|
||||
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
|
||||
# here. Per invocation, not per container — authoring a browser task shouldn't need a
|
||||
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
|
||||
# still mirrors an ordinary trial.
|
||||
#
|
||||
# The correction must be appended AFTER the base note, which says there is no Read tool.
|
||||
RACCOON_TOOLS="Bash"
|
||||
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
|
||||
RACCOON_TOOLS="Bash,Read"
|
||||
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
|
||||
|
||||
\$(cat "$read_note_src")"
|
||||
fi
|
||||
export RACCOON_TOOLS
|
||||
# Only mention the browser on an image that actually has one — most don't. Probed at
|
||||
# launch, not baked in, so the same launcher is correct in whichever container it runs.
|
||||
#
|
||||
# Exported two ways because the harnesses take extra instructions differently: claude
|
||||
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
|
||||
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
|
||||
# (it has no str_replace_editor), so the browser part is exported on its own too.
|
||||
RACCOON_BROWSER_NOTE=""
|
||||
RACCOON_BROWSER_FLAGS=()
|
||||
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
|
||||
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
|
||||
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
|
||||
|
||||
\${RACCOON_BROWSER_NOTE}"
|
||||
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
|
||||
fi
|
||||
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
|
||||
export RACCOON_HARNESS="$id"
|
||||
# These launchers exist only in explore, and a refresh that has to CREATE a config
|
||||
# needs the surface to know the capture hooks belong in it.
|
||||
export RACCOON_SURFACE=explore
|
||||
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
|
||||
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
|
||||
# — it has no CLAUDE_PLUGIN_* to fall back to. Exporting it ALSO overrode the dir for
|
||||
# claude, whose slash command pins --plugin-data to the plugin dir, so the hook wrote one
|
||||
# place and capture read another and the recorded session was silently ignored.
|
||||
$refresh_line
|
||||
$launch
|
||||
LAUNCHER
|
||||
chmod +x "$bin/raccoon-explore-$cli"
|
||||
echo "harness-setup: launcher raccoon-explore-$cli" >&2
|
||||
done < <(_harness_query --explore-launchers 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Alias lines for ~/.bashrc.
|
||||
harness_alias_lines() {
|
||||
local id cli launch switchable
|
||||
local browser_clis=""
|
||||
while IFS=$'\t' read -r id cli launch; do
|
||||
[ -n "$cli" ] && [ -n "$launch" ] || continue
|
||||
echo "alias $cli=\"raccoon-explore-$cli\""
|
||||
# Same derivation as the launcher: only a harness whose launch line takes
|
||||
# $RACCOON_TOOLS has a toolset the flag can change.
|
||||
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
|
||||
browser_clis="${browser_clis:+$browser_clis }$cli"
|
||||
fi
|
||||
done < <(_harness_query --explore-launchers 2>/dev/null || true)
|
||||
|
||||
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
|
||||
# alternate screen buffer, so anything printed just before exec is hidden for the whole
|
||||
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
|
||||
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
|
||||
#
|
||||
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
|
||||
# and in one without.
|
||||
[ -n "$browser_clis" ] || return 0
|
||||
local first="${browser_clis%% *}"
|
||||
cat <<HINT
|
||||
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
|
||||
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
|
||||
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
|
||||
fi
|
||||
HINT
|
||||
}
|
||||
|
||||
# Write each harness's config file from the registry, replacing whatever was there.
|
||||
#
|
||||
# The file is OWNED, not merged: TOML has no way to return to the document root after a
|
||||
# table header, so appending or prepending around foreign content silently reparents
|
||||
# root-level keys into whichever table happens to precede them. Owning it also means a
|
||||
# registry change actually reaches a container that was already set up.
|
||||
harness_write_configs() {
|
||||
local id config_path blob target tmp
|
||||
while IFS=$'\t' read -r id config_path blob; do
|
||||
[ -n "$config_path" ] && [ -n "$blob" ] || continue
|
||||
# Guarded: a bare failing assignment exits the caller's `set -e` post-create with
|
||||
# no explanation. A path this cannot expand is one harness's problem, not the
|
||||
# container's.
|
||||
target=$(eval "printf '%s' \"$config_path\"") || {
|
||||
echo "harness-setup: WARNING $id config_path could not be expanded — skipping" >&2
|
||||
continue
|
||||
}
|
||||
mkdir -p "$(dirname "$target")"
|
||||
tmp="$target.raccoon-tmp"
|
||||
# Expansion is strict: an unset var would otherwise be written through as the
|
||||
# literal ${VAR}, which surfaces much later as an unparseable value.
|
||||
if ! {
|
||||
echo "# Generated from harness-registry.toml — edits here are overwritten."
|
||||
printf '%s' "$blob" | base64 -d | python3 -c '
|
||||
import os, re, sys
|
||||
text = sys.stdin.read()
|
||||
missing = sorted(
|
||||
{m.group(1) for m in re.finditer(r"\$\{(\w+)\}", text) if m.group(1) not in os.environ}
|
||||
)
|
||||
if missing:
|
||||
sys.stderr.write("unset: " + ", ".join(missing) + "\n")
|
||||
raise SystemExit(1)
|
||||
sys.stdout.write(os.path.expandvars(text))
|
||||
'
|
||||
} > "$tmp"; then
|
||||
rm -f "$tmp"
|
||||
echo "harness-setup: WARNING $id config NOT written — a value it needs is unset." >&2
|
||||
echo "harness-setup: run harness_setup_credentials first (harness_setup_all does)." >&2
|
||||
continue
|
||||
fi
|
||||
mv "$tmp" "$target"
|
||||
echo "harness-setup: $id config -> $target" >&2
|
||||
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Link every available skill into each harness's skills_dir, for harnesses that declare one.
|
||||
# Both container layouts are covered: the explore container holds the snapshot skill under
|
||||
# plugins/, the authoring container holds the authoring skills under .claude/skills. Whichever
|
||||
# directories exist here are the ones this container has.
|
||||
harness_install_skills() {
|
||||
local sources="${RACCOON_SKILL_SOURCE_DIRS:-/workspace/plugins/create-snapshot/skills /workspace/.claude/skills}"
|
||||
local id dir target src skill name installed
|
||||
while IFS=$'\t' read -r id dir; do
|
||||
[ -n "$dir" ] || continue
|
||||
target=$(eval "printf '%s' \"$dir\"") || {
|
||||
echo "harness-setup: WARNING $id skills_dir could not be expanded — skipping" >&2
|
||||
continue
|
||||
}
|
||||
mkdir -p "$target"
|
||||
installed=0
|
||||
for src in $sources; do
|
||||
[ -d "$src" ] || continue
|
||||
for skill in "$src"/*/; do
|
||||
[ -f "$skill/SKILL.md" ] || continue
|
||||
name=$(basename "$skill")
|
||||
ln -sfn "${skill%/}" "$target/$name"
|
||||
installed=$((installed + 1))
|
||||
done
|
||||
done
|
||||
echo "harness-setup: $id skills -> $target ($installed linked)" >&2
|
||||
done < <(_harness_query --skills-dirs 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# The lines that explain a setup failure are printed as it happens, and the devcontainer
|
||||
# CLI's own stack trace lands on top of them. Close with a banner so the worker has
|
||||
# something to look for, and something to send us.
|
||||
_harness_fatal_banner() {
|
||||
echo "" >&2
|
||||
echo " ============================================================" >&2
|
||||
echo " HARNESS SETUP FAILED — this container has no agent CLI." >&2
|
||||
echo "" >&2
|
||||
echo " The harness-setup: lines above say why. Anything the" >&2
|
||||
echo " devcontainer prints after this is a consequence, not the" >&2
|
||||
echo " cause; send us the harness-setup: lines." >&2
|
||||
echo " ============================================================" >&2
|
||||
echo "" >&2
|
||||
}
|
||||
|
||||
harness_setup_all() {
|
||||
harness_preflight || { _harness_fatal_banner; return 1; }
|
||||
harness_setup_credentials
|
||||
harness_write_auth
|
||||
harness_install_clis
|
||||
harness_write_configs
|
||||
harness_install_skills
|
||||
# Launchers are NOT installed here. They are an Explore concern (that container aliases
|
||||
# `claude`/`codex` to them), and it passes its own AGENT_CLI_DIR — installing them here
|
||||
# too wrote every launcher twice, the first time with the wrong editor path, and left an
|
||||
# unused one in the authoring container.
|
||||
echo "harness-setup: authoring harnesses" >&2
|
||||
harness_report || { _harness_fatal_banner; return 1; }
|
||||
}
|
||||
837
worker-toolkit-potion-polyglot-orig/scripts/snapshot-to-task.ts
Normal file
837
worker-toolkit-potion-polyglot-orig/scripts/snapshot-to-task.ts
Normal file
@@ -0,0 +1,837 @@
|
||||
/**
|
||||
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
|
||||
*/
|
||||
|
||||
import { execFileSync, execSync } from 'child_process';
|
||||
import {
|
||||
chmodSync,
|
||||
copyFileSync,
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
readFileSync,
|
||||
readdirSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'fs';
|
||||
import { basename, join, resolve } from 'path';
|
||||
import pino from 'pino';
|
||||
import pinoPretty from 'pino-pretty';
|
||||
import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { copyTree } from './lib/copy-tree';
|
||||
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
|
||||
|
||||
// --- CLI ---
|
||||
|
||||
const argv = yargs(hideBin(process.argv))
|
||||
.option('snapshot', {
|
||||
type: 'string',
|
||||
describe: 'Path to the snapshot directory',
|
||||
demandOption: true,
|
||||
})
|
||||
.option('json', {
|
||||
type: 'boolean',
|
||||
describe: 'Output structured JSON logs',
|
||||
default: false,
|
||||
})
|
||||
.strict()
|
||||
.help()
|
||||
.parseSync();
|
||||
|
||||
const log = pino(
|
||||
{ name: 'snapshot-to-task', level: 'info' },
|
||||
argv.json
|
||||
? process.stdout
|
||||
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
|
||||
);
|
||||
|
||||
// --- Read snapshot data ---
|
||||
|
||||
const snapshotDir = argv.snapshot;
|
||||
|
||||
if (!existsSync(snapshotDir)) {
|
||||
log.fatal(
|
||||
{ path: snapshotDir },
|
||||
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
interface SnapshotMetadata {
|
||||
slug: string;
|
||||
session_uuid: string;
|
||||
/** Absent on snapshots captured before harness selection existed. */
|
||||
harness?: string;
|
||||
original_cwd: string;
|
||||
commit: string | null;
|
||||
branch: string | null;
|
||||
remote_url: string | null;
|
||||
timestamp: string;
|
||||
plugin_version: string;
|
||||
}
|
||||
|
||||
interface Annotation {
|
||||
what_trying: string;
|
||||
what_hoping: string;
|
||||
what_happened: string;
|
||||
[key: string]: string;
|
||||
}
|
||||
|
||||
const metadata = JSON.parse(
|
||||
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
|
||||
) as SnapshotMetadata;
|
||||
const annotation = JSON.parse(
|
||||
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
|
||||
) as Annotation;
|
||||
|
||||
if (!metadata.slug) {
|
||||
log.fatal(
|
||||
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const slug = metadata.slug;
|
||||
|
||||
// --- Locate harbor infrastructure ---
|
||||
|
||||
function findRepoRoot(): string | null {
|
||||
let dir = process.cwd();
|
||||
while (dir !== resolve(dir, '..')) {
|
||||
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
|
||||
dir = resolve(dir, '..');
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const maybeRepoRoot = findRepoRoot();
|
||||
|
||||
if (!maybeRepoRoot) {
|
||||
log.fatal(
|
||||
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const repoRoot: string = maybeRepoRoot;
|
||||
|
||||
const harborTasks = join(repoRoot, 'harbor-tasks');
|
||||
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
|
||||
const sharedDir = sharedCandidates.find((d) => existsSync(d));
|
||||
const taskDir = join(harborTasks, slug);
|
||||
|
||||
if (existsSync(taskDir)) {
|
||||
log.fatal(
|
||||
{ path: taskDir },
|
||||
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (!sharedDir) {
|
||||
log.fatal(
|
||||
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// --- Detect repo name ---
|
||||
|
||||
interface ToolkitConfig {
|
||||
repo: string;
|
||||
defaultCommit: string;
|
||||
/** The packed kit's release version (git describe at pack time). */
|
||||
version?: string;
|
||||
}
|
||||
|
||||
function readToolkitConfig(): ToolkitConfig | null {
|
||||
const configPath = join(repoRoot, 'toolkit.json');
|
||||
if (!existsSync(configPath)) return null;
|
||||
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
|
||||
}
|
||||
|
||||
function repoNameFromRemote(remoteUrl: string | null): string | null {
|
||||
if (!remoteUrl) return null;
|
||||
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
|
||||
return match ? match[1] : null;
|
||||
}
|
||||
|
||||
function findSubmoduleDir(remoteUrl: string | null): string | null {
|
||||
if (!remoteUrl) return null;
|
||||
const reposDir = join(repoRoot, 'repos');
|
||||
if (!existsSync(reposDir)) return null;
|
||||
|
||||
const normalize = (url: string) =>
|
||||
url
|
||||
.replace(/\.git$/, '')
|
||||
.replace(/^git@github\.com:/, 'https://github.com/')
|
||||
.toLowerCase();
|
||||
|
||||
for (const entry of readdirSync(reposDir)) {
|
||||
const repoPath = join(reposDir, entry, 'repo');
|
||||
if (!existsSync(repoPath)) continue;
|
||||
try {
|
||||
const remote = execSync('git remote get-url origin', {
|
||||
cwd: repoPath,
|
||||
encoding: 'utf8',
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
}).trim();
|
||||
if (normalize(remote) === normalize(remoteUrl)) return entry;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const toolkitConfig = readToolkitConfig();
|
||||
|
||||
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
|
||||
// Derive which member this task targets from the snapshot's original_cwd basename,
|
||||
// validated against the member list.
|
||||
const polyglotMember = (() => {
|
||||
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
|
||||
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
|
||||
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
|
||||
const members = cfg.repos.map((r) => r.repo);
|
||||
return base && members.includes(base) ? base : null;
|
||||
})();
|
||||
const repoName =
|
||||
polyglotMember ??
|
||||
toolkitConfig?.repo ??
|
||||
findSubmoduleDir(metadata.remote_url) ??
|
||||
repoNameFromRemote(metadata.remote_url);
|
||||
|
||||
if (!repoName) {
|
||||
log.fatal(
|
||||
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
|
||||
const sessionUuid = metadata.session_uuid;
|
||||
|
||||
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
|
||||
|
||||
// --- Create task directory structure ---
|
||||
|
||||
mkdirSync(join(taskDir, 'environment'), { recursive: true });
|
||||
mkdirSync(join(taskDir, 'tests'), { recursive: true });
|
||||
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
|
||||
|
||||
// --- Copy shared infrastructure ---
|
||||
|
||||
// The complete grader asset set test.sh depends on: the grader system prompt
|
||||
// and the renderer (test.sh exits without the renderer). Sources missing from
|
||||
// task-shared/ are skipped by the existsSync guard below.
|
||||
const sharedFiles = [
|
||||
{ src: 'test.sh', dest: 'tests/test.sh' },
|
||||
{
|
||||
src: 'grader-system-prompt-consolidated.md',
|
||||
dest: 'tests/grader-system-prompt-consolidated.md',
|
||||
},
|
||||
// test.sh execs this to render the grade; without it the verifier writes no reward
|
||||
// file and the trial errors out rather than scoring.
|
||||
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
|
||||
];
|
||||
|
||||
for (const { src, dest } of sharedFiles) {
|
||||
const srcPath = join(sharedDir, src);
|
||||
const destPath = join(taskDir, dest);
|
||||
if (existsSync(srcPath)) {
|
||||
copyFileSync(srcPath, destPath);
|
||||
if (src === 'test.sh') chmodSync(destPath, 0o755);
|
||||
log.debug({ src, dest }, 'Copied shared file');
|
||||
} else {
|
||||
log.warn({ src }, 'Shared file not found');
|
||||
}
|
||||
}
|
||||
|
||||
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
|
||||
// their output to the grader as evidence for the CORRECTNESS score, so without
|
||||
// them a code task's correctness is never signal-backed — the grader falls back
|
||||
// to reading the diff alone. Same per-member-then-generic resolution as the
|
||||
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
|
||||
// member, a single-repo toolkit ships the lone test-commands.sh.
|
||||
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
|
||||
const genericTestCommands = join(sharedDir, 'test-commands.sh');
|
||||
const testCommandsSrc = existsSync(perMemberTestCommands)
|
||||
? perMemberTestCommands
|
||||
: genericTestCommands;
|
||||
if (existsSync(testCommandsSrc)) {
|
||||
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
|
||||
copyFileSync(testCommandsSrc, testCommandsDest);
|
||||
chmodSync(testCommandsDest, 0o755);
|
||||
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
|
||||
} else {
|
||||
// Not fatal: the grader still scores correctness by walking the changed code.
|
||||
log.info(
|
||||
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
|
||||
);
|
||||
}
|
||||
|
||||
// --- Write Dockerfile with session resume support ---
|
||||
//
|
||||
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
|
||||
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
|
||||
// staging COPY/RUN steps. Session staging happens after the original CMD —
|
||||
// COPY and RUN are layer ops independent of CMD, so the original
|
||||
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
|
||||
//
|
||||
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
|
||||
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
|
||||
|
||||
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
|
||||
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
|
||||
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
|
||||
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
|
||||
const taskSharedDockerfile = existsSync(perMemberDockerfile)
|
||||
? perMemberDockerfile
|
||||
: join(repoRoot, 'task-shared', 'Dockerfile');
|
||||
let baseDockerfile: string;
|
||||
if (existsSync(taskSharedDockerfile)) {
|
||||
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
|
||||
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
|
||||
} else {
|
||||
log.warn(
|
||||
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
|
||||
);
|
||||
baseDockerfile = `FROM debian:bookworm-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y \\
|
||||
git \\
|
||||
python3 \\
|
||||
curl \\
|
||||
jq \\
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install Claude Code globally (needed by the grader in test.sh)
|
||||
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
|
||||
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
|
||||
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
|
||||
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
|
||||
|
||||
WORKDIR /workspace
|
||||
COPY workspace/ .
|
||||
|
||||
# Block network tools — agent should only read code and write documents
|
||||
RUN mkdir -p .claude && \\
|
||||
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
|
||||
|
||||
RUN git init && \\
|
||||
git config user.email "dev@agent" && \\
|
||||
git config user.name "Dev" && \\
|
||||
git add -A && \\
|
||||
git commit -m "initial" --quiet
|
||||
|
||||
CMD ["sleep", "infinity"]
|
||||
`;
|
||||
}
|
||||
|
||||
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
|
||||
// toolkit's own append rather than an edit to the Dockerfile.
|
||||
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
|
||||
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
|
||||
// directories in the context, so the layer errors with `"/session": not found`.
|
||||
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
|
||||
// files are copied into environment/, so the task-side copy is not there yet.
|
||||
const sessionSiblingDir = join(snapshotDir, 'session');
|
||||
const hasSessionSibling =
|
||||
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
|
||||
|
||||
const sessionStaging = `
|
||||
# >>> toolkit-managed: snapshot-session >>>
|
||||
# Stage session files for the snapshot agent adapter to install at runtime.
|
||||
COPY session.jsonl /tmp/snapshot-session/session.jsonl
|
||||
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
|
||||
# <<< toolkit-managed <<<
|
||||
`;
|
||||
|
||||
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
|
||||
|
||||
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
|
||||
log.debug('Wrote Dockerfile (per-repo base + session staging)');
|
||||
|
||||
// --- Copy snapshot.patch as workspace.patch ---
|
||||
|
||||
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
|
||||
if (existsSync(snapshotPatch)) {
|
||||
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
|
||||
log.debug('Copied snapshot.patch -> workspace.patch');
|
||||
}
|
||||
|
||||
// --- Scrub the worker's filesystem layout out of the session ---
|
||||
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
|
||||
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
|
||||
|
||||
const WORKSPACE_MOUNT = '/workspace';
|
||||
|
||||
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
|
||||
/** Member names when this toolkit is polyglot; empty means single-repo. */
|
||||
const MEMBER_NAMES: readonly string[] = (() => {
|
||||
const dir = join(repoRoot, 'repos');
|
||||
if (!existsSync(dir)) return [];
|
||||
try {
|
||||
return readdirSync(dir, { withFileTypes: true })
|
||||
.filter((e) => e.isDirectory())
|
||||
.map((e) => e.name);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
})();
|
||||
|
||||
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
|
||||
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
|
||||
function repoRootOf(cwd: string): string | null {
|
||||
if (MEMBER_NAMES.length > 0) {
|
||||
// A real member of THIS toolkit wins; the generic shape covers a member whose
|
||||
// directory the toolkit no longer has (an older snapshot, a renamed member).
|
||||
for (const name of MEMBER_NAMES) {
|
||||
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
|
||||
if (hit) return hit[1];
|
||||
}
|
||||
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
|
||||
if (generic) return generic[1];
|
||||
}
|
||||
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
|
||||
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
|
||||
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
|
||||
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
|
||||
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
|
||||
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
|
||||
// and a session spanning two checkouts is scrubbed rather than skipped.
|
||||
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
|
||||
.filter((r): r is string => r !== null)
|
||||
.sort((a, b) => b.length - a.length);
|
||||
const { sanitized } = sanitizeSessionJsonl(raw, {
|
||||
cwdPrefixes: roots,
|
||||
placeholder: WORKSPACE_MOUNT,
|
||||
});
|
||||
return { text: sanitized, roots };
|
||||
}
|
||||
|
||||
// --- Copy session files for --resume ---
|
||||
//
|
||||
// The full session.jsonl (including any post-end_turn entries) goes into the
|
||||
// task root for reference. A truncated version — keeping everything up to
|
||||
// and including the last assistant entry with stop_reason="end_turn" — goes
|
||||
// into environment/ for the container. Stopping on a clean assistant turn
|
||||
// avoids Claude Code's synthetic "No response requested." injection when
|
||||
// the session is resumed with --fork-session and a new --print prompt.
|
||||
|
||||
const sessionJsonl = join(snapshotDir, 'session.jsonl');
|
||||
if (existsSync(sessionJsonl)) {
|
||||
// Fail-open: a session this can't scrub ships exactly as it was, because a
|
||||
// leaked path is a smaller problem than a task that can't be created.
|
||||
let sessionText = readFileSync(sessionJsonl, 'utf8');
|
||||
try {
|
||||
const { text, roots } = scrubWorkerPaths(sessionText);
|
||||
if (roots.length > 0) {
|
||||
sessionText = text;
|
||||
log.info(
|
||||
{ roots, mountedAt: WORKSPACE_MOUNT },
|
||||
'Rewrote the authoring checkout path to the trial mount point'
|
||||
);
|
||||
} else {
|
||||
log.debug('No worker-rooted cwd to rewrite; session used as-is');
|
||||
}
|
||||
} catch (err) {
|
||||
log.warn(
|
||||
{ err: err instanceof Error ? err.message : String(err) },
|
||||
'Could not rewrite paths in the session; using it as-is'
|
||||
);
|
||||
}
|
||||
|
||||
// Full version for reference
|
||||
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
|
||||
log.debug('Wrote full session.jsonl to task root');
|
||||
|
||||
// Truncated version for the container: strip everything from the last
|
||||
// user text turn onwards. This drops the failure-eliciting question
|
||||
// (which `--print` will redeliver to the trial agent as the new prompt)
|
||||
// AND the failure response itself (so the trial agent doesn't see its
|
||||
// previous answer), while preserving conversational context up to the
|
||||
// last clean assistant `end_turn`.
|
||||
//
|
||||
// Algorithm (refined Option B):
|
||||
// 1. Find U = index of the last user-text turn that is NOT a slash
|
||||
// command (use the same command-marker filter as
|
||||
// extractLastUserMessage).
|
||||
// 2. Walk backwards from U - 1 to find the last `assistant` entry
|
||||
// with stop_reason: "end_turn".
|
||||
// 3. Truncate slice(0, lastEndTurnIndex + 1).
|
||||
//
|
||||
// If U doesn't exist or no end_turn assistant precedes U, write an
|
||||
// empty session.jsonl — the snapshot agent adapter detects this and
|
||||
// skips --resume entirely, starting fresh from --print.
|
||||
const sessionLines = sessionText.trimEnd().split('\n');
|
||||
|
||||
// A non-Claude session is not a Claude transcript, so the scan below finds no
|
||||
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
|
||||
// applies the same rule in that harness's own format.
|
||||
const harness = metadata.harness ?? 'claude-code';
|
||||
const isClaude = harness === 'claude-code';
|
||||
|
||||
let lastUserTextIndex = -1;
|
||||
for (let i = 0; i < sessionLines.length; i++) {
|
||||
try {
|
||||
const entry = JSON.parse(sessionLines[i]) as {
|
||||
type?: string;
|
||||
isCompactSummary?: boolean;
|
||||
message?: { content?: unknown };
|
||||
};
|
||||
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
|
||||
// Compaction summaries are synthetic user turns whose text often quotes
|
||||
// earlier /create-snapshot:snapshot runs — never the command turn itself,
|
||||
// so they must not trip the break below.
|
||||
if (entry.isCompactSummary) continue;
|
||||
const content = entry.message.content;
|
||||
// Mirror extractLastUserMessage: skip the snapshot command itself
|
||||
// and any slash-command / local-command marker turns.
|
||||
if (content.includes('create-snapshot:snapshot')) break;
|
||||
if (
|
||||
content.includes('<command-name>') ||
|
||||
content.includes('<command-message>') ||
|
||||
content.includes('<local-command-caveat>')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
lastUserTextIndex = i;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
let lastEndTurnIndex = -1;
|
||||
if (lastUserTextIndex > 0) {
|
||||
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
|
||||
try {
|
||||
const entry = JSON.parse(sessionLines[i]) as {
|
||||
type?: string;
|
||||
message?: { stop_reason?: unknown };
|
||||
};
|
||||
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
|
||||
lastEndTurnIndex = i;
|
||||
break;
|
||||
}
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!isClaude) {
|
||||
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
|
||||
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
|
||||
const truncated = stripAuthoringScaffolding(harness, kept);
|
||||
writeFileSync(
|
||||
join(taskDir, 'environment', 'session.jsonl'),
|
||||
truncated.length ? truncated.join('\n') + '\n' : ''
|
||||
);
|
||||
log.debug(
|
||||
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
|
||||
'Wrote truncated session.jsonl to environment/ (harness reader)'
|
||||
);
|
||||
} else if (lastEndTurnIndex >= 0) {
|
||||
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
|
||||
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
|
||||
log.debug(
|
||||
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
|
||||
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
|
||||
);
|
||||
} else {
|
||||
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
|
||||
if (lastUserTextIndex < 0) {
|
||||
log.warn(
|
||||
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
|
||||
);
|
||||
} else {
|
||||
log.warn(
|
||||
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
const sessionDir = join(snapshotDir, 'session');
|
||||
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
|
||||
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
|
||||
// Claude Code writes subagent files write-only (--w-------). Fix them so
|
||||
// Harbor's dirhash can read them during environment setup.
|
||||
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
|
||||
log.debug('Copied session/');
|
||||
} else {
|
||||
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
|
||||
}
|
||||
|
||||
// The harness that captured the snapshot; the trial runs this one.
|
||||
const harness =
|
||||
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
|
||||
|
||||
/**
|
||||
* The model and effort this harness defaulted to when the task was authored, recorded
|
||||
* for reference only — nothing reads these back, and a trial still resolves both from
|
||||
* the registry at run time. Best-effort: a task is not worth failing over a note.
|
||||
*/
|
||||
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
|
||||
try {
|
||||
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
|
||||
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
|
||||
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
|
||||
// registry needs tomllib. Best-effort, so a miss just omits the note.
|
||||
let python = '';
|
||||
for (const candidate of [
|
||||
process.env.RACCOON_PYTHON,
|
||||
'python3',
|
||||
'python3.13',
|
||||
'python3.12',
|
||||
'python3.11',
|
||||
]) {
|
||||
if (!candidate) continue;
|
||||
try {
|
||||
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
|
||||
python = candidate;
|
||||
break;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (!python) return null;
|
||||
const rows = execFileSync(python, [resolver, '--defaults'], {
|
||||
encoding: 'utf-8',
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
});
|
||||
for (const line of rows.split('\n')) {
|
||||
const [id, model, effort] = line.split('\t');
|
||||
if (id === harnessId && model) return { model, effort: effort ?? '' };
|
||||
}
|
||||
} catch {
|
||||
// registry unreadable here — omit the note
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const authored = authoredDefaults(harness);
|
||||
|
||||
// --- Write task.toml ---
|
||||
|
||||
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
|
||||
// so there's nothing to set here.
|
||||
const taskToml = `version = "1.0"
|
||||
|
||||
[metadata]
|
||||
program = "raccoon"
|
||||
author = "rl-env-coding"
|
||||
category = "sdlc/technical-writing"
|
||||
repo = "${repoName}"
|
||||
commit = "${commitShort}"
|
||||
# The toolkit release this task was created with. Written by the toolkit —
|
||||
# leave it in place: task tooling reads it to know which toolkit's assets
|
||||
# this task grades with.
|
||||
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
|
||||
snapshot = "${basename(snapshotDir)}"
|
||||
session_uuid = "${sessionUuid}"
|
||||
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
|
||||
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
|
||||
browser = false
|
||||
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
|
||||
|
||||
[verifier]
|
||||
timeout_sec = 7200.0
|
||||
|
||||
[agent]
|
||||
harness = "${harness}"
|
||||
timeout_sec = 18000.0
|
||||
|
||||
[environment]
|
||||
build_timeout_sec = 6000.0
|
||||
cpus = 2
|
||||
memory_mb = 4096
|
||||
storage_mb = 10240
|
||||
gpus = 0
|
||||
allow_internet = true
|
||||
|
||||
[verifier.env]
|
||||
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
|
||||
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
|
||||
|
||||
[solution.env]
|
||||
`;
|
||||
|
||||
writeFileSync(join(taskDir, 'task.toml'), taskToml);
|
||||
log.debug('Wrote task.toml');
|
||||
|
||||
// --- Extract instruction from session transcript ---
|
||||
|
||||
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
|
||||
if (!existsSync(sessionPath)) return null;
|
||||
|
||||
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
|
||||
|
||||
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
|
||||
// and the worker silently gets a placeholder instruction. Its reader applies the same
|
||||
// rule — last real user turn, ignoring command invocations — in that harness's format.
|
||||
if (harness !== 'claude-code') {
|
||||
const userTurns = turnsFromLines(harness, lines).filter(
|
||||
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
|
||||
);
|
||||
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
|
||||
}
|
||||
|
||||
let lastUserMessage: string | null = null;
|
||||
|
||||
for (const line of lines) {
|
||||
try {
|
||||
const entry = JSON.parse(line) as {
|
||||
type?: string;
|
||||
isCompactSummary?: boolean;
|
||||
message?: { content?: unknown };
|
||||
};
|
||||
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
|
||||
// Synthetic compaction summary — not a real user turn, and its text
|
||||
// often quotes earlier /create-snapshot:snapshot runs.
|
||||
if (entry.isCompactSummary) continue;
|
||||
const content = entry.message.content;
|
||||
if (content.includes('create-snapshot:snapshot')) break;
|
||||
if (
|
||||
content.includes('<command-name>') ||
|
||||
content.includes('<command-message>') ||
|
||||
content.includes('<local-command-caveat>')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
lastUserMessage = content;
|
||||
}
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return lastUserMessage;
|
||||
}
|
||||
|
||||
const lastUserMessage = extractLastUserMessage(
|
||||
join(snapshotDir, 'session.jsonl'),
|
||||
metadata.harness ?? 'claude-code'
|
||||
);
|
||||
|
||||
const instructionHeader =
|
||||
'# Replace this with your refined task instruction\n\n' +
|
||||
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
|
||||
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
|
||||
|
||||
if (lastUserMessage) {
|
||||
writeFileSync(
|
||||
join(taskDir, 'instruction.md'),
|
||||
instructionHeader + lastUserMessage.trimEnd() + '\n'
|
||||
);
|
||||
log.info('Wrote instruction.md (from last user message in session)');
|
||||
} else {
|
||||
writeFileSync(
|
||||
join(taskDir, 'instruction.md'),
|
||||
instructionHeader +
|
||||
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
|
||||
);
|
||||
log.warn('Could not extract instruction from session — needs manual editing');
|
||||
}
|
||||
|
||||
// --- Scaffold holistic-rubric.md ---
|
||||
|
||||
const holisticRubricMd = `<!--
|
||||
HOLISTIC RUBRIC — the file trials grade against. Run
|
||||
/write-holistic-rubric
|
||||
to draft it interactively, or point Claude Code at this file,
|
||||
session-full.jsonl, and task-shared/grading-standard.md.
|
||||
|
||||
Snapshot: ${basename(snapshotDir)}
|
||||
Session: ${metadata.session_uuid}
|
||||
Repo: ${metadata.remote_url}
|
||||
Commit: ${metadata.commit}
|
||||
|
||||
## What happened in the snapshot conversation
|
||||
|
||||
The worker was trying to: ${annotation.what_trying}
|
||||
They hoped Claude would: ${annotation.what_hoping}
|
||||
Instead, Claude: ${annotation.what_happened}
|
||||
|
||||
## What this file contains
|
||||
|
||||
The eight-criterion Grading Standard
|
||||
(task-shared/grading-standard.md, embedded in
|
||||
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
|
||||
Correctness, Broader Correctness / craft, Persistence, Communication,
|
||||
Verification & Thoroughness, Common Sense, and Thought Partnership. This
|
||||
file adds the task-specific knowledge the grader cannot infer: full task
|
||||
context, the ground truth you established, what strong and weak responses
|
||||
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
|
||||
fraction subtractions with a named criterion target, never points, never
|
||||
caps. The document must stand alone: the grader sees only it and the
|
||||
shared standard.
|
||||
-->
|
||||
|
||||
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
|
||||
including the instructions above. -->
|
||||
`;
|
||||
|
||||
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
|
||||
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
|
||||
|
||||
// --- Build workspace ---
|
||||
|
||||
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
|
||||
|
||||
if (existsSync(buildScript)) {
|
||||
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
|
||||
try {
|
||||
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
|
||||
cwd: repoRoot,
|
||||
encoding: 'utf8',
|
||||
stdio: 'inherit',
|
||||
// build-workspace does a bulk-file write burst (git archive|tar of the
|
||||
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
|
||||
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
|
||||
// falls back to a full copy across filesystems). On a slow bind mount
|
||||
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
|
||||
// path) that legitimately runs into minutes, so a tight cap false-fails a
|
||||
// working-but-slow build as "not runnable". Keep this generous — it's only
|
||||
// a backstop against a true hang; the real Harbor build downstream budgets
|
||||
// build_timeout_sec = 6000.
|
||||
timeout: 1_200_000,
|
||||
});
|
||||
} catch (e: unknown) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
|
||||
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
|
||||
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
|
||||
process.exit(1);
|
||||
}
|
||||
} else {
|
||||
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
try {
|
||||
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
|
||||
stdio: 'ignore',
|
||||
timeout: 5000,
|
||||
});
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
|
||||
// --- Done ---
|
||||
|
||||
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
|
||||
log.info('Next steps:');
|
||||
log.info(' 1. Review instruction.md');
|
||||
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
|
||||
log.info(' 3. Run calibration trials to validate scoring tiers');
|
||||
989
worker-toolkit-potion-polyglot-orig/scripts/snapshot_agent.py
Normal file
989
worker-toolkit-potion-polyglot-orig/scripts/snapshot_agent.py
Normal file
@@ -0,0 +1,989 @@
|
||||
"""
|
||||
Harbor agent adapters built on the stock claude-code adapter.
|
||||
|
||||
- ``PreinstalledClaudeCode``: stock behavior, except agent-setup reuses the
|
||||
claude binary baked into the task image instead of re-downloading it.
|
||||
- ``SnapshotClaudeCode``: extends it to inject --resume and --fork-session
|
||||
when a snapshot session is present in the environment.
|
||||
|
||||
The harbor-run script uses PreinstalledClaudeCode for manual tasks and
|
||||
auto-detects snapshot-based tasks to use SnapshotClaudeCode.
|
||||
"""
|
||||
|
||||
import base64
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import shutil
|
||||
import tarfile
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import atif_session
|
||||
import browser_note
|
||||
try:
|
||||
from dnsjail import apply_dns_jail
|
||||
except ImportError: # no helper shipped -> no jail, rather than no trials
|
||||
|
||||
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
|
||||
return None
|
||||
|
||||
from harbor.agents.installed.base import CliFlag
|
||||
from harbor.agents.installed.claude_code import ClaudeCode
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
|
||||
_UUID_FILE = "/tmp/snapshot-session/uuid.txt"
|
||||
_SESSION_FILE = "/tmp/snapshot-session/session.jsonl"
|
||||
_DISABLED_SUFFIX = ".snapshot-seeded-disabled"
|
||||
|
||||
# The "[1m]" model-id suffix is a benchmark convention for the 1M-context
|
||||
# window, NOT a real model id — the wire request must send the plain id plus
|
||||
# this beta header (the proxy's server-side alias for the suffixed id was
|
||||
# dropped; the literal id now 400s and the claude CLI hangs retrying).
|
||||
_CONTEXT_1M_BETA_HEADER = "anthropic-beta: context-1m-2025-08-07"
|
||||
|
||||
|
||||
def _strip_1m_suffix(model: str) -> tuple[str, bool]:
|
||||
"""Split a model id into (wire id, wants-1M-context). The [1m] tag stays in
|
||||
agent identity (name(), trial-config model_name); only the wire id drops it."""
|
||||
if model.endswith("[1m]"):
|
||||
return model[: -len("[1m]")], True
|
||||
return model, False
|
||||
|
||||
|
||||
def _merge_custom_headers(*values: str | None) -> str:
|
||||
"""Union newline-separated ANTHROPIC_CUSTOM_HEADERS values, keeping order.
|
||||
Local, not shared: this module ships to the toolkit without llm_proxy_env."""
|
||||
merged: list[str] = []
|
||||
for value in values:
|
||||
for line in (value or "").split("\n"):
|
||||
if line.strip() and line not in merged:
|
||||
merged.append(line)
|
||||
return "\n".join(merged)
|
||||
|
||||
|
||||
def _with_1m_beta_header(existing: str | None) -> str:
|
||||
"""Merge the 1M-context beta header into an ANTHROPIC_CUSTOM_HEADERS value."""
|
||||
return _merge_custom_headers(existing, _CONTEXT_1M_BETA_HEADER)
|
||||
|
||||
# Fast mode (claude's /fast), opted in per-trial via `--ak fast_mode=true`
|
||||
# (harbor-run --fast). The --settings flag is the only headless opt-in, and it
|
||||
# doubles as the availability override for a proxy base URL: claude probes
|
||||
# org fast-mode status at an API route inference-only proxies don't forward,
|
||||
# and without the flag that failed probe reads as "unavailable". Rendered
|
||||
# pre-quoted because a bool CliFlag emits its `cli` string verbatim into a
|
||||
# shell command.
|
||||
_FAST_MODE_CLI = "--settings " + shlex.quote('{"fastMode":true}')
|
||||
|
||||
_log = logging.getLogger("snapshot-agent")
|
||||
|
||||
# --- Reduced "bash + str_replace_editor" tool surface (the CANONICAL agent) --
|
||||
# The canonical agent runs with ONLY the built-in Bash tool (so there's no
|
||||
# async-MCP startup race), and a str_replace_editor file editor is delivered as a
|
||||
# CLI it invokes through Bash. The editor's logic is vendored verbatim under
|
||||
# scripts/str_replace_editor_vendor/ and wrapped by scripts/str_replace_editor.
|
||||
# We stage both into the sandbox at install time and point the agent at them via
|
||||
# --append-system-prompt.
|
||||
#
|
||||
# The toolset is chosen by WHICH AGENT CLASS harbor runs, not by an env var: the
|
||||
# reduced toolset is the canonical PreinstalledClaudeCode / SnapshotClaudeCode;
|
||||
# the full Claude Code built-in toolset (Read/Edit/Write/Grep/...) is the SEPARATE,
|
||||
# transitional FullToolsetPreinstalledClaudeCode / FullToolsetSnapshotClaudeCode
|
||||
# (delete those once every snapshot session.jsonl is recorded in the reduced
|
||||
# format). A different toolset is simply a different agent — see name() below.
|
||||
_AGENT_CLI_DIR = "/opt/agent-cli"
|
||||
_AGENT_CLI_BIN = f"{_AGENT_CLI_DIR}/str_replace_editor"
|
||||
# Files copied (orchestrator-relative) into the sandbox tar, arcname -> source.
|
||||
_AGENT_CLI_FILES = {
|
||||
"str_replace_editor_vendor/__init__.py": "str_replace_editor_vendor/__init__.py",
|
||||
"str_replace_editor_vendor/base.py": "str_replace_editor_vendor/base.py",
|
||||
"str_replace_editor_vendor/run.py": "str_replace_editor_vendor/run.py",
|
||||
"str_replace_editor_vendor/edit.py": "str_replace_editor_vendor/edit.py",
|
||||
"str_replace_editor": "str_replace_editor",
|
||||
}
|
||||
_AGENT_CLI_NOTE_FALLBACK = (
|
||||
"You are running with a restricted toolset: your ONLY built-in tool is Bash.\n\n"
|
||||
"To view and edit files, use the `str_replace_editor` command-line tool (it "
|
||||
"replicates the standard str_replace-based file editor). Invoke it from Bash "
|
||||
f"by piping ONE JSON object to {_AGENT_CLI_BIN} on stdin. Use a quoted "
|
||||
"heredoc so backslashes and quotes are preserved:\n\n"
|
||||
f" {_AGENT_CLI_BIN} <<'EDITOR'\n"
|
||||
' {"command":"view","path":"/abs/path/file.rb"}\n'
|
||||
" EDITOR\n\n"
|
||||
"Commands (the JSON \"command\" field):\n"
|
||||
"- view: view a file (optionally add \"view_range\":[start,end]) or list a directory.\n"
|
||||
"- create: create a NEW file -> {\"command\":\"create\",\"path\":...,\"file_text\":\"...\"} (fails if it already exists).\n"
|
||||
"- str_replace: replace a UNIQUE substring -> {\"command\":\"str_replace\",\"path\":...,\"old_str\":\"...\",\"new_str\":\"...\"}.\n"
|
||||
"- insert: insert at a line -> {\"command\":\"insert\",\"path\":...,\"insert_line\":N,\"insert_text\":\"...\"}.\n\n"
|
||||
"All paths must be absolute. JSON strings must be valid (escape newlines as \\n "
|
||||
"and double-quotes as \\\"). For everything else (running commands, searching "
|
||||
"with grep/find, reading via sed, etc.) use Bash directly."
|
||||
)
|
||||
|
||||
|
||||
def _toolset_note(with_browser: bool = False, with_read: bool = False) -> str:
|
||||
"""The toolset note appended to Claude Code's stock ``--print`` system prompt
|
||||
(via ``--append-system-prompt``) for the canonical reduced toolset.
|
||||
|
||||
We APPEND rather than replace: the stock prompt's big block carries the
|
||||
DYNAMIC Environment section (cwd, platform, OS, model) generated per run, and
|
||||
``--system-prompt`` (full replace) would drop it — leaving the reduced-toolset
|
||||
agent without the Environment/Memory sections the full-toolset agent has (and
|
||||
those differ host-vs-sandbox, so they can't be hardcoded faithfully).
|
||||
Appending keeps the sandbox's real block intact; this note is added last to
|
||||
override the two stock spots that name tools this harness lacks (the "prefer
|
||||
the dedicated file/search tools" Harness bullet and the Memory section's "use
|
||||
the Write tool"). Single source of truth is toolset_note.md (read as-is, with
|
||||
only surrounding whitespace trimmed). Falls back to the built-in note if the
|
||||
file is missing."""
|
||||
path = Path(__file__).resolve().parent / "toolset_note.md"
|
||||
try:
|
||||
note = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
note = _AGENT_CLI_NOTE_FALLBACK
|
||||
# Order matters: the Read correction must come AFTER the base note, because it supersedes
|
||||
# that note's "there are no Read/Grep/Glob tools" line. Shipping the base note alone to a
|
||||
# Read-enabled agent would be a false statement about its own toolset.
|
||||
if with_read:
|
||||
read_note = Path(__file__).resolve().parent / "toolset_note_read.md"
|
||||
try:
|
||||
note = f"{note}\n\n{read_note.read_text(encoding='utf-8').strip()}"
|
||||
except OSError:
|
||||
_log.warning("toolset_note_read.md missing; Read correction omitted")
|
||||
if with_browser:
|
||||
extra = browser_note.browser_note()
|
||||
if extra:
|
||||
note = f"{note}\n\n{extra}"
|
||||
return note
|
||||
|
||||
|
||||
class PreinstalledClaudeCode(ClaudeCode):
|
||||
"""Canonical agent: the reduced ``bash + str_replace_editor`` toolset, with
|
||||
agent-setup reusing the claude binary baked into the task image instead of
|
||||
re-downloading it. Manual (non-snapshot) tasks use this directly.
|
||||
|
||||
A different toolset is a different AGENT (not an env-var flag), so this class
|
||||
names itself ``claude-code-reduced-toolset`` — the agent identity carries the
|
||||
toolset and there's no separate toolset field anywhere. The full Claude Code
|
||||
built-in toolset is the transitional :class:`FullToolsetPreinstalledClaudeCode`
|
||||
(name ``claude-code``). (Harbor records the name in each trial's result.json;
|
||||
we don't use ``harbor traces export`` — which would otherwise require a
|
||||
registry name — so a descriptive non-registry name is fine. Provenance is also
|
||||
in the trial config's ``agent.import_path``.)
|
||||
"""
|
||||
|
||||
# Set by _probe_browser() during install(); read by build_cli_flags(). Declared
|
||||
# here so the full-toolset subclass (which skips the probe) still has a value.
|
||||
_has_browser = False
|
||||
|
||||
# Adding fast_mode HERE (the shared base) makes `--ak fast_mode=true` a known
|
||||
# kwarg on every Claude agent class, and every build_cli_flags variant —
|
||||
# reduced, browser, full-toolset — renders it via this table.
|
||||
CLI_FLAGS = [
|
||||
*ClaudeCode.CLI_FLAGS,
|
||||
CliFlag("fast_mode", cli=_FAST_MODE_CLI, type="bool"),
|
||||
]
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "claude-code-reduced-toolset"
|
||||
|
||||
# Manual tasks inherit harbor's run(), so the jail has to be applied here as well as on
|
||||
# the snapshot path — which overrides run() and never reaches this one.
|
||||
async def run(self, instruction, environment, context) -> None: # type: ignore[override]
|
||||
await apply_dns_jail(self, environment)
|
||||
await super().run(instruction, environment, context)
|
||||
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
# Normalize a "[1m]"-suffixed model id HERE, in the shared base of all
|
||||
# four Claude agent classes, so every run path sends the plain wire id —
|
||||
# including stock ClaudeCode.run(), which this class inherits for manual
|
||||
# (non-snapshot) tasks and which reads self.model_name directly. The 1M
|
||||
# window is requested via the beta header instead, delivered through
|
||||
# harbor's extra_env channel (merged into every agent exec on 0.9.x,
|
||||
# wired via Trial.scoped_exec_env on 0.18.x); _build_env also mirrors it
|
||||
# for the snapshot run path. The [1m] identity survives on purpose:
|
||||
# BaseAgent's _init_model_info already cached the suffixed id (for
|
||||
# to_agent_info) before this rebinding, and the trial config's
|
||||
# agent.model_name records the id as passed on the CLI.
|
||||
model = self.model_name
|
||||
if not model:
|
||||
# Stock ClaudeCode.run() falls back to os.environ["ANTHROPIC_MODEL"]
|
||||
# verbatim, with no overridable hook on the manual-task path — so
|
||||
# when no model was pinned, adopt a [1m]-suffixed env model here
|
||||
# (same effective wire value, normalized). A plain env model stays
|
||||
# on the stock fallback path untouched.
|
||||
model = os.environ.get("ANTHROPIC_MODEL", "")
|
||||
stripped, wants_1m = _strip_1m_suffix(model)
|
||||
if wants_1m:
|
||||
self.model_name = stripped
|
||||
self._extra_env["ANTHROPIC_CUSTOM_HEADERS"] = _with_1m_beta_header(
|
||||
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS")
|
||||
)
|
||||
|
||||
async def install(self, environment) -> None:
|
||||
"""Canonical (reduced) setup: stage the str_replace_editor CLI, then ensure
|
||||
the claude binary. The full-toolset subclass skips the staging (it has no
|
||||
editor CLI) and reuses ``_ensure_claude_binary`` directly."""
|
||||
await self._stage_agent_cli(environment)
|
||||
await self._ensure_claude_binary(environment)
|
||||
await self._probe_browser(environment)
|
||||
|
||||
async def _probe_browser(self, environment) -> None:
|
||||
"""Record whether this image ships the Playwright `pw` wrapper, so the toolset
|
||||
note mentions the browser only on images that have one. Runs during install(),
|
||||
which harbor calls before build_cli_flags() reads the result. The probe and the
|
||||
note text are shared with the codex adapter via browser_note.py."""
|
||||
self._has_browser = await browser_note.probe_browser(environment)
|
||||
|
||||
async def _ensure_claude_binary(self, environment) -> None:
|
||||
"""Reuse the claude binary already baked into the task image instead
|
||||
of re-downloading it at agent-setup.
|
||||
|
||||
Harbor's stock ``install()`` pipes ``claude.ai/install.sh`` to bash
|
||||
inside the live sandbox — a ~240 MB binary download racing the 360 s
|
||||
agent-setup timeout. On a slow or stalling egress path (classic on
|
||||
WSL2/Docker Desktop) the download never finishes and every trial dies
|
||||
with ``AgentSetupTimeoutError`` — pure waste, since our task
|
||||
Dockerfiles already bake claude into ``/usr/local/bin``. Probe for a
|
||||
working binary first; fall back to harbor's installer only when the
|
||||
image truly lacks one, when an explicit agent-version pin doesn't
|
||||
match the baked binary, or when the probe itself errors. The probe
|
||||
is local and sub-second, so the fallbacks are effectively no worse
|
||||
than stock behavior.
|
||||
"""
|
||||
try:
|
||||
probe = await environment.exec(
|
||||
command=(
|
||||
'export PATH="$HOME/.local/bin:$PATH"; '
|
||||
"command -v claude >/dev/null 2>&1 && claude --version"
|
||||
),
|
||||
timeout_sec=30,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — any probe failure → stock path
|
||||
_log.debug(
|
||||
"claude preinstall probe failed (%s); using stock installer", exc
|
||||
)
|
||||
await ClaudeCode.install(self, environment)
|
||||
return
|
||||
|
||||
if probe.return_code == 0:
|
||||
pinned = getattr(self, "_version", None)
|
||||
baked = self.parse_version(probe.stdout or "")
|
||||
if not pinned or baked == pinned:
|
||||
_log.info(
|
||||
"claude already in image (%s); skipping runtime download",
|
||||
(probe.stdout or "").strip(),
|
||||
)
|
||||
return
|
||||
_log.info(
|
||||
"image bakes claude %s but %s was requested; using stock installer",
|
||||
baked,
|
||||
pinned,
|
||||
)
|
||||
await ClaudeCode.install(self, environment)
|
||||
|
||||
def build_cli_flags(self) -> str:
|
||||
"""Emit the reduced ``bash + str_replace_editor`` toolset flags: restrict
|
||||
the built-in toolset to ``--tools Bash`` and append the toolset note.
|
||||
|
||||
harbor's stock adapter exposes only ``--allowedTools`` /
|
||||
``--disallowedTools`` (permission lists). Under
|
||||
``--permission-mode=bypassPermissions`` (which both run paths use) an
|
||||
allowlist does NOT remove tools — every built-in stays available, just
|
||||
auto-approved. claude's ``--tools`` flag is the one that sets the
|
||||
AVAILABLE toolset; ``Bash`` leaves Bash as the only built-in.
|
||||
|
||||
APPEND (not replace) the toolset note: --append keeps the sandbox's real,
|
||||
dynamically-generated Environment/Memory block intact, and the note (added
|
||||
last) overrides the stock prompt's references to tools this harness lacks
|
||||
(the "prefer the dedicated file/search tools" bullet and the Memory
|
||||
section's "use the Write tool"). See _toolset_note(). The full-toolset
|
||||
subclass overrides this back to stock ``ClaudeCode.build_cli_flags``.
|
||||
"""
|
||||
flags = super().build_cli_flags()
|
||||
note = _toolset_note(with_browser=self._has_browser)
|
||||
extra = f"--tools Bash --append-system-prompt {shlex.quote(note)}"
|
||||
return f"{flags} {extra}" if flags else extra
|
||||
|
||||
async def _claude_format_session_path(self, environment, env, session_uuid: str) -> str:
|
||||
"""Path in the sandbox to a session.jsonl Claude can resume.
|
||||
|
||||
A task authored on another harness stages THAT harness's native blob, which
|
||||
`claude --resume` cannot read. Convert it through the ATIF hub and upload the
|
||||
result. A Claude-authored task — every task before multi-harness authoring —
|
||||
returns the staged path untouched, so its install stays byte-identical.
|
||||
"""
|
||||
async def _read(command: str) -> str | None:
|
||||
try:
|
||||
result = await environment.exec(command=command, env=env, timeout_sec=30)
|
||||
except Exception as exc: # best-effort: fall back to installing as-is
|
||||
_log.warning("Could not read staged session (%s); installing verbatim", exc)
|
||||
return None
|
||||
return getattr(result, "stdout", "") or ""
|
||||
|
||||
# Probe the head first: a Claude-authored session needs no conversion, and that is
|
||||
# the common case, so pulling a multi-megabyte transcript through exec to learn
|
||||
# only that is waste. Act on the probe only when it says "claude" — a truncated
|
||||
# ATIF blob (one JSON object) parses as nothing, which is not the same answer.
|
||||
head = await _read(f"head -c 8192 {_SESSION_FILE} 2>/dev/null || true")
|
||||
if head is None:
|
||||
return _SESSION_FILE
|
||||
if atif_session.detect_format(head) == "claude":
|
||||
return _SESSION_FILE
|
||||
|
||||
text = await _read(f"cat {_SESSION_FILE} 2>/dev/null || true")
|
||||
if text is None:
|
||||
return _SESSION_FILE
|
||||
fmt = atif_session.detect_format(text)
|
||||
if fmt in (None, "claude"):
|
||||
return _SESSION_FILE
|
||||
|
||||
from datetime import datetime, timezone
|
||||
|
||||
iso_ts = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z")
|
||||
lines = atif_session.atif_to_claude_session(
|
||||
atif_session.to_atif(text), session_id=session_uuid, iso_ts=iso_ts
|
||||
)
|
||||
if not lines:
|
||||
_log.warning("Converted %s session was empty; installing verbatim", fmt)
|
||||
return _SESSION_FILE
|
||||
|
||||
_log.info(
|
||||
"Seeding Claude from a %s-authored session: %d records via ATIF", fmt, len(lines)
|
||||
)
|
||||
remote = "/tmp/snapshot-session/session.claude.jsonl"
|
||||
with tempfile.NamedTemporaryFile(
|
||||
"w", suffix=".jsonl", delete=False, encoding="utf-8"
|
||||
) as tmp:
|
||||
tmp.write("\n".join(lines) + "\n")
|
||||
host_path = tmp.name
|
||||
try:
|
||||
await environment.upload_file(host_path, remote)
|
||||
if environment.default_user is not None:
|
||||
await self.exec_as_root(
|
||||
environment, command=f"chown {environment.default_user} {shlex.quote(remote)}"
|
||||
)
|
||||
finally:
|
||||
try:
|
||||
os.unlink(host_path)
|
||||
except OSError:
|
||||
pass
|
||||
return remote
|
||||
|
||||
async def _stage_agent_cli(self, environment) -> None:
|
||||
"""Stage the vendored str_replace_editor CLI into the sandbox.
|
||||
|
||||
Bundles scripts/str_replace_editor_vendor/ (the verbatim EditTool source)
|
||||
plus the wrapper into a tar, ships it as base64 in one root exec, and
|
||||
unpacks it to ``/opt/agent-cli`` with the wrapper made executable. Runs at
|
||||
install time so the CLI is present before the agent's first turn.
|
||||
"""
|
||||
here = Path(__file__).resolve().parent
|
||||
buf = io.BytesIO()
|
||||
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
|
||||
for arcname, rel in _AGENT_CLI_FILES.items():
|
||||
src = here / rel
|
||||
if not src.is_file():
|
||||
raise FileNotFoundError(
|
||||
f"str_replace_editor: missing vendored file {src} "
|
||||
f"(expected scripts/{rel})"
|
||||
)
|
||||
tar.add(str(src), arcname=arcname)
|
||||
b64 = base64.b64encode(buf.getvalue()).decode()
|
||||
command = (
|
||||
f"mkdir -p {_AGENT_CLI_DIR} && "
|
||||
f"printf %s {shlex.quote(b64)} | base64 -d | "
|
||||
f"tar xzf - -C {_AGENT_CLI_DIR} && "
|
||||
f"chmod -R a+rX {_AGENT_CLI_DIR} && chmod a+rx {_AGENT_CLI_BIN} && "
|
||||
# Fail loudly at setup if python3 is absent — the CLI needs it.
|
||||
f"command -v python3 >/dev/null 2>&1 || "
|
||||
f'{{ echo "str_replace_editor: python3 not found in sandbox" >&2; exit 1; }}'
|
||||
)
|
||||
await self.exec_as_root(environment, command=command, timeout_sec=120)
|
||||
_log.info("Staged str_replace_editor CLI to %s", _AGENT_CLI_BIN)
|
||||
await self._selftest_agent_cli(environment)
|
||||
|
||||
async def _selftest_agent_cli(self, environment) -> None:
|
||||
"""Fail loudly at setup if the staged CLI can't actually run an edit.
|
||||
|
||||
Invokes the REAL staged wrapper (``_AGENT_CLI_BIN``) exactly the way the
|
||||
agent will — one JSON object piped on stdin — and asserts the edit landed
|
||||
on disk. This exercises the whole path end-to-end (the wrapper's shebang,
|
||||
its exec permissions, stdin JSON parsing, ``sys.path`` into the vendored
|
||||
package, and the vendored EditTool itself), so a broken wrapper, wrong
|
||||
perms, or a bad python (the vendored edit.py uses PEP 604 ``X | None`` and
|
||||
needs python >= 3.10) errors the trial at agent-setup instead of silently
|
||||
mid-benchmark. The python version is logged for visibility.
|
||||
|
||||
``set -e`` + the trailing OK echo mean any failure in the pipe (the
|
||||
wrapper) or the grep yields rc != 0 / no OK marker, which we raise on."""
|
||||
json_fmt = (
|
||||
'{"command":"str_replace","path":"%s",'
|
||||
'"old_str":"alpha","new_str":"ALPHA"}'
|
||||
)
|
||||
cmd = (
|
||||
"python3 --version 2>&1; "
|
||||
"set -e; "
|
||||
'TMP="$(mktemp)"; '
|
||||
'printf "alpha\\nbeta\\n" > "$TMP"; '
|
||||
f"printf '{json_fmt}' \"$TMP\" | {_AGENT_CLI_BIN}; "
|
||||
'grep -q ALPHA "$TMP"; '
|
||||
'echo "str_replace_editor self-test OK"'
|
||||
)
|
||||
result = await self.exec_as_root(environment, command=cmd, timeout_sec=30)
|
||||
out = (getattr(result, "stdout", "") or "").strip()
|
||||
rc = getattr(result, "return_code", 0)
|
||||
_log.info("str_replace_editor self-test (rc=%s): %s", rc, out.replace("\n", " | "))
|
||||
if rc != 0 or "self-test OK" not in out:
|
||||
raise RuntimeError(
|
||||
f"str_replace_editor self-test failed (rc={rc}). The staged "
|
||||
f"str_replace_editor could not perform an edit in the sandbox "
|
||||
f"(often python < 3.10). Output:\n{out}"
|
||||
)
|
||||
|
||||
|
||||
class SnapshotClaudeCode(PreinstalledClaudeCode):
|
||||
"""Canonical snapshot agent: resumes from a snapshot session and runs the
|
||||
reduced ``bash + str_replace_editor`` toolset. The full Claude Code built-in
|
||||
toolset is the transitional :class:`FullToolsetSnapshotClaudeCode`."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "snapshot-claude-code-reduced-toolset"
|
||||
|
||||
@staticmethod
|
||||
def _is_bedrock_mode() -> bool:
|
||||
return False
|
||||
|
||||
async def run(self, instruction: str, environment, context) -> None:
|
||||
env = self._build_env()
|
||||
config_dir = env["CLAUDE_CONFIG_DIR"]
|
||||
|
||||
# Read the snapshot session UUID from the container
|
||||
result = await environment.exec(
|
||||
command=f"cat {_UUID_FILE} 2>/dev/null || echo ''",
|
||||
env=env,
|
||||
timeout_sec=5,
|
||||
)
|
||||
session_uuid = result.stdout.strip() if result.stdout else ""
|
||||
|
||||
# Check whether the staged session.jsonl has any meaningful content.
|
||||
# snapshot-to-task may write an empty file when the snapshot has no
|
||||
# assistant entry with stop_reason="end_turn" — in that case we must
|
||||
# NOT pass --resume / --fork-session (CC errors out on an empty
|
||||
# session) and we must NOT stage the empty file.
|
||||
session_has_content = False
|
||||
if session_uuid:
|
||||
size_result = await environment.exec(
|
||||
command=(
|
||||
f"if [ -s {_SESSION_FILE} ] && "
|
||||
f"grep -q '[^[:space:]]' {_SESSION_FILE} 2>/dev/null; "
|
||||
f"then echo 'yes'; else echo 'no'; fi"
|
||||
),
|
||||
env=env,
|
||||
timeout_sec=5,
|
||||
)
|
||||
session_has_content = (
|
||||
size_result.stdout.strip() == "yes" if size_result.stdout else False
|
||||
)
|
||||
|
||||
install_src = _SESSION_FILE
|
||||
escaped_instruction = shlex.quote(instruction)
|
||||
cli_flags = self.build_cli_flags()
|
||||
extra_flags = (cli_flags + " ") if cli_flags else ""
|
||||
|
||||
# Install the session JSONL from the container's filesystem (COPY'd
|
||||
# in by the Dockerfile) into Claude Code's config dir. This runs in
|
||||
# the same exec call as the claude command so files are visible.
|
||||
workspace_dir = f"{config_dir}/projects/-workspace"
|
||||
if session_uuid and session_has_content:
|
||||
resume_flags = f"--resume {session_uuid} --fork-session "
|
||||
seeded_jsonl = f"{workspace_dir}/{session_uuid}.jsonl"
|
||||
install_src = await self._claude_format_session_path(
|
||||
environment, env, session_uuid
|
||||
)
|
||||
install_prefix = (
|
||||
f'mkdir -p "{workspace_dir}" && '
|
||||
f'cp "{install_src}" "{seeded_jsonl}" && '
|
||||
f'chmod -R 777 "{config_dir}" && '
|
||||
)
|
||||
# After the run, drop the seeded session JSONL so harbor's
|
||||
# trajectory converter sees ONLY the forked session claude wrote.
|
||||
# ``--fork-session`` writes the forked conversation (a SUPERSET: it
|
||||
# copies the seeded history verbatim, reusing the same ``toolu_*``
|
||||
# ids) to a NEW ``{uuid}.jsonl``. If the seed is left behind,
|
||||
# harbor's ``_convert_events_to_trajectory`` globs BOTH files and
|
||||
# merges them, producing duplicate ``tool_result`` events; the
|
||||
# second one orphans a tool_call (empty ``tool_name``), which
|
||||
# ``_convert_event_to_step`` skips, leaving a gap that trips the
|
||||
# sequential ``step_id`` invariant on the ``Trajectory`` model — so
|
||||
# the whole conversion raises and ``trajectory.json`` never lands.
|
||||
# Removing the seed in-sandbox makes a snapshot trial look exactly
|
||||
# like a stock claude-code trial (one session file) and is robust
|
||||
# even when the host-side ``populate_context_post_run`` hook below
|
||||
# is bypassed (e.g. a stale agent module on harbor's import path).
|
||||
# Guard: only remove the seed if a *different* forked JSONL exists,
|
||||
# so a claude build that appended in place (no real fork) keeps its
|
||||
# sole session file.
|
||||
seeded_cleanup = "; " + self._seeded_cleanup_cmd(
|
||||
workspace_dir, session_uuid
|
||||
)
|
||||
else:
|
||||
resume_flags = ""
|
||||
install_prefix = ""
|
||||
seeded_cleanup = ""
|
||||
if session_uuid and not session_has_content:
|
||||
# Seeded session.jsonl is empty (no end_turn assistant in the
|
||||
# source snapshot). Start fresh with --print instead.
|
||||
_log.debug(
|
||||
"Seeded session.jsonl is empty; skipping --resume and starting fresh"
|
||||
)
|
||||
|
||||
await apply_dns_jail(self, environment)
|
||||
await self.exec_as_agent(
|
||||
environment,
|
||||
command=(
|
||||
f'{install_prefix}'
|
||||
f'export PATH="$HOME/.local/bin:$PATH"; '
|
||||
f'export CLAUDE_CONFIG_DIR="{config_dir}"; '
|
||||
f"claude --verbose --output-format=stream-json "
|
||||
f"--permission-mode=bypassPermissions "
|
||||
f"{resume_flags}"
|
||||
f"{extra_flags}"
|
||||
f"--print -- {escaped_instruction} 2>&1 </dev/null | tee "
|
||||
f"/logs/agent/claude-code.txt"
|
||||
f"{seeded_cleanup}"
|
||||
),
|
||||
env=env,
|
||||
)
|
||||
|
||||
def _build_env(self) -> dict[str, str]:
|
||||
"""Build the environment dict for agent execution."""
|
||||
env: dict[str, str | None] = {
|
||||
"ANTHROPIC_API_KEY": os.environ.get("ANTHROPIC_API_KEY", ""),
|
||||
"ANTHROPIC_BASE_URL": os.environ.get("ANTHROPIC_BASE_URL", None),
|
||||
"CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
|
||||
"IS_SANDBOX": "1",
|
||||
"FORCE_AUTO_BACKGROUND_TASKS": "1",
|
||||
"ENABLE_BACKGROUND_TASKS": "1",
|
||||
}
|
||||
|
||||
# The host env's header list (e.g. the proxy's X-Project-Id entry from
|
||||
# apply_llm_proxy_env) rides along with any agent-requested headers —
|
||||
# but only on a proxy base URL: a project id must never reach a
|
||||
# provider's own API (the direct-API escape hatch).
|
||||
on_proxy = "/llm_proxy/" in (env["ANTHROPIC_BASE_URL"] or "")
|
||||
header = _merge_custom_headers(
|
||||
os.environ.get("ANTHROPIC_CUSTOM_HEADERS") if on_proxy else None,
|
||||
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS"),
|
||||
)
|
||||
if self.model_name:
|
||||
# "[1m]" normalization happened once in PreinstalledClaudeCode.__init__
|
||||
# (shared by all Claude agent classes); by here model_name is the plain
|
||||
# wire id and any 1M beta header sits in self._extra_env.
|
||||
env["ANTHROPIC_MODEL"] = self.model_name.split("/")[-1]
|
||||
elif "ANTHROPIC_MODEL" in os.environ:
|
||||
# A [1m] env model was adopted into model_name by __init__, so this
|
||||
# fallback normally only sees plain ids — but the env can change
|
||||
# after construction, so normalize here too (defense in depth).
|
||||
fallback, wants_1m = _strip_1m_suffix(os.environ["ANTHROPIC_MODEL"])
|
||||
if wants_1m:
|
||||
header = _with_1m_beta_header(header)
|
||||
env["ANTHROPIC_MODEL"] = fallback
|
||||
|
||||
# Mirror the [1m] beta header into this env dict: harbor versions differ
|
||||
# in where extra_env is merged (0.9.x: per-exec in _exec; 0.18.x:
|
||||
# Trial-level scoped_exec_env), so carrying it here keeps the snapshot
|
||||
# run path correct regardless of which plumbing the installed harbor has.
|
||||
if header:
|
||||
env["ANTHROPIC_CUSTOM_HEADERS"] = header
|
||||
|
||||
if os.environ.get("CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING", "").strip() == "1":
|
||||
env["CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING"] = "1"
|
||||
|
||||
env.update(self._resolved_env_vars)
|
||||
env["CLAUDE_CONFIG_DIR"] = (EnvironmentPaths.agent_dir / "sessions").as_posix()
|
||||
|
||||
return {k: v for k, v in env.items() if v}
|
||||
|
||||
@staticmethod
|
||||
def _seeded_cleanup_cmd(workspace_dir: str, session_uuid: str) -> str:
|
||||
"""Shell snippet (run after the claude pipeline) that deletes the seeded
|
||||
session JSONL so harbor's converter sees ONLY the forked session.
|
||||
|
||||
Guarded: the seed is removed only if a *different* ``*.jsonl`` exists in
|
||||
``workspace_dir`` — i.e. claude actually forked to a new file. If claude
|
||||
appended in place (no real fork), the seed is the sole session file and
|
||||
is kept, so we never destroy the only record of the run.
|
||||
"""
|
||||
seeded_jsonl = f"{workspace_dir}/{session_uuid}.jsonl"
|
||||
return (
|
||||
f'if [ -f "{seeded_jsonl}" ] && '
|
||||
f'ls "{workspace_dir}/"*.jsonl 2>/dev/null '
|
||||
f'| grep -vq "/{session_uuid}\\.jsonl$"; '
|
||||
f'then rm -f "{seeded_jsonl}"; fi'
|
||||
)
|
||||
|
||||
def populate_context_post_run(self, context) -> None:
|
||||
"""Override harbor's post-run hook to make trajectory.json production
|
||||
reliable for snapshot resumes.
|
||||
|
||||
The seeded session JSONL (the file we cp'd in from
|
||||
``/tmp/snapshot-session/session.jsonl`` during run-prep) lives in
|
||||
``sessions/projects/-workspace/{seeded_uuid}.jsonl`` alongside the
|
||||
new forked-session JSONL claude actually wrote during this trial.
|
||||
Harbor's ``_convert_events_to_trajectory`` reads BOTH files, and
|
||||
events from the seeded session — often from an older claude version
|
||||
with a slightly different schema — trip skip-paths inside
|
||||
``_convert_event_to_step``. Skipping events breaks the sequential
|
||||
``step_id`` invariant the ``Trajectory`` pydantic model enforces, so
|
||||
the whole conversion errors out and ``trajectory.json`` never lands.
|
||||
|
||||
We work around it by moving every JSONL whose stem is not the
|
||||
forked session id aside before delegating to the parent's hook,
|
||||
then restoring it after. The forked session id is read from
|
||||
``claude-code.txt``'s ``system/init`` event, which is the first
|
||||
thing claude writes via ``--output-format=stream-json``.
|
||||
|
||||
If, after our cleanup, trajectory.json still isn't there, we log
|
||||
at WARNING (not debug) so downstream consumers can see something
|
||||
went wrong rather than silently inherit a half-broken run.
|
||||
"""
|
||||
moved = self._isolate_forked_session_jsonl()
|
||||
try:
|
||||
super().populate_context_post_run(context)
|
||||
finally:
|
||||
self._restore_moved_jsonls(moved)
|
||||
|
||||
trajectory_path = self.logs_dir / "trajectory.json"
|
||||
if not trajectory_path.is_file():
|
||||
# Stock conversion produced nothing even from the isolated forked
|
||||
# session. The usual culprit is a single orphaned tool_result (e.g.
|
||||
# left behind by autocompaction) that harbor skips, leaving a
|
||||
# step_id gap that sinks the whole Trajectory. Recover by rebuilding
|
||||
# from the forked session with orphaned tool_results stripped — one
|
||||
# degenerate event should not cost us the entire trajectory.
|
||||
if self._recover_trajectory_stripping_orphans():
|
||||
self.logger.info(
|
||||
"Recovered trajectory.json by stripping orphaned tool_results"
|
||||
)
|
||||
|
||||
if not trajectory_path.is_file():
|
||||
self.logger.warning(
|
||||
"trajectory.json was NOT produced in %s. Downstream tooling "
|
||||
"(grader replay, worldbench export, publish-reference-runs) "
|
||||
"depends on it. claude-code.txt: %s. sessions/projects: %s",
|
||||
self.logs_dir,
|
||||
"present" if (self.logs_dir / "claude-code.txt").is_file() else "MISSING",
|
||||
self._summarize_sessions_dir(),
|
||||
)
|
||||
|
||||
def _read_forked_session_id(self) -> str | None:
|
||||
"""Return the ``session_id`` from the first ``system/init`` event in
|
||||
``claude-code.txt``, or None if it can't be found.
|
||||
|
||||
Claude Code's ``--output-format=stream-json --print`` mode emits the
|
||||
init event near the top of the stream, but is allowed to emit other
|
||||
framing events before it (provider notices, warnings, etc.). Scan
|
||||
every line until we find an init event with a usable session_id, or
|
||||
we hit EOF — don't bail on the first non-init JSON we see.
|
||||
"""
|
||||
stream_path = self.logs_dir / "claude-code.txt"
|
||||
if not stream_path.is_file():
|
||||
return None
|
||||
try:
|
||||
with open(stream_path, "r", encoding="utf-8") as handle:
|
||||
for line in handle:
|
||||
stripped = line.strip()
|
||||
if not stripped or not stripped.startswith("{"):
|
||||
continue
|
||||
try:
|
||||
event = json.loads(stripped)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if (
|
||||
event.get("type") == "system"
|
||||
and event.get("subtype") == "init"
|
||||
):
|
||||
sid = event.get("session_id")
|
||||
if isinstance(sid, str) and sid:
|
||||
return sid
|
||||
except OSError:
|
||||
return None
|
||||
return None
|
||||
|
||||
def _recover_trajectory_stripping_orphans(self) -> bool:
|
||||
"""Last-resort rebuild of ``trajectory.json`` from the forked session,
|
||||
tolerant of the single degenerate event that harbor's converter would
|
||||
otherwise let sink the whole trajectory.
|
||||
|
||||
Harbor assigns ``step_id`` from the enumerate index BEFORE it may skip an
|
||||
event, so any event that ``_convert_event_to_step`` raises on leaves a
|
||||
gap that fails ``Trajectory.validate_step_ids`` — and the whole
|
||||
conversion is lost. Two real shapes trigger this even in a single,
|
||||
already-isolated forked session:
|
||||
|
||||
- an orphaned ``tool_result`` whose ``tool_use`` was summarized away by
|
||||
autocompaction; and
|
||||
- a ``tool_result`` that shares an identical timestamp with its
|
||||
``tool_use`` and stable-sorts ahead of it, so the result is processed
|
||||
before the call exists (also yielding an empty ``tool_name``).
|
||||
|
||||
We re-run harbor's own conversion but temporarily make
|
||||
``_convert_event_to_step`` substitute a placeholder step instead of
|
||||
raising, so one bad event costs us that single observation rather than
|
||||
the entire run. Returns ``True`` iff ``trajectory.json`` was written.
|
||||
This runs ONLY after the stock conversion already failed, so it never
|
||||
changes behavior on healthy sessions.
|
||||
"""
|
||||
sessions_root = self.logs_dir / "sessions" / "projects"
|
||||
if not sessions_root.is_dir():
|
||||
return False
|
||||
jsonls: list[Path] = []
|
||||
for project_dir in sessions_root.iterdir():
|
||||
if project_dir.is_dir():
|
||||
jsonls.extend(project_dir.glob("*.jsonl"))
|
||||
if not jsonls:
|
||||
return False
|
||||
|
||||
# Prefer the forked session alone; fall back to whatever is present.
|
||||
forked = self._read_forked_session_id()
|
||||
if forked and any(j.stem == forked for j in jsonls):
|
||||
jsonls = [j for j in jsonls if j.stem == forked]
|
||||
|
||||
from harbor.models.trajectories.step import Step
|
||||
|
||||
original_convert = self._convert_event_to_step
|
||||
|
||||
def tolerant_convert(event: dict, step_id: int) -> Step:
|
||||
try:
|
||||
return original_convert(event, step_id)
|
||||
except ValueError:
|
||||
# Degenerate tool event (orphaned / mis-ordered tool_result).
|
||||
# Keep its output as a user observation so nothing is silently
|
||||
# dropped, and the sequential step_id stays intact.
|
||||
output = event.get("output")
|
||||
call_id = event.get("call_id") or "?"
|
||||
message = (
|
||||
output
|
||||
if isinstance(output, str) and output.strip()
|
||||
else f"[unmatched tool_result for {call_id}]"
|
||||
)
|
||||
# Guard the timestamp: Step.validate_timestamp raises ValueError
|
||||
# on a non-ISO-8601 value, which harbor's loop would catch and
|
||||
# skip — re-introducing the exact step_id gap we're recovering
|
||||
# from. Fall back to no timestamp rather than lose the step.
|
||||
ts = event.get("timestamp")
|
||||
try:
|
||||
return Step(
|
||||
step_id=step_id, timestamp=ts, source="user", message=message
|
||||
)
|
||||
except ValueError:
|
||||
return Step(
|
||||
step_id=step_id, timestamp=None, source="user", message=message
|
||||
)
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
session_dir = Path(tmp) / "-workspace"
|
||||
session_dir.mkdir(parents=True)
|
||||
for jsonl in jsonls:
|
||||
shutil.copy(jsonl, session_dir / jsonl.name)
|
||||
self._convert_event_to_step = tolerant_convert # type: ignore[assignment]
|
||||
try:
|
||||
trajectory = self._convert_events_to_trajectory(session_dir)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
self.logger.debug("Tolerant recovery failed: %s", exc)
|
||||
return False
|
||||
finally:
|
||||
del self._convert_event_to_step
|
||||
if not trajectory:
|
||||
return False
|
||||
try:
|
||||
with open(self.logs_dir / "trajectory.json", "w", encoding="utf-8") as handle:
|
||||
json.dump(
|
||||
trajectory.to_json_dict(), handle, indent=2, ensure_ascii=False
|
||||
)
|
||||
except OSError:
|
||||
return False
|
||||
return True
|
||||
|
||||
def _isolate_forked_session_jsonl(self) -> list[tuple[Path, Path]]:
|
||||
"""Move any JSONL not matching the forked session id to a sibling
|
||||
``*.snapshot-seeded-disabled`` path. Returns the list of
|
||||
(original, disabled) pairs so they can be restored after.
|
||||
|
||||
If we can't determine the forked session id (no claude-code.txt, no
|
||||
init event, etc.), we leave the directory untouched. Harbor's
|
||||
converter will run as today — if it succeeds, great; if not, our
|
||||
WARNING fires.
|
||||
|
||||
Safety guardrail: if NO JSONL on disk matches the forked id (e.g.
|
||||
claude wrote to an unexpected path), don't move anything aside —
|
||||
that would leave the converter with an empty session dir and
|
||||
guarantee failure. Better to let harbor's normal flow attempt the
|
||||
conversion against what's actually there.
|
||||
"""
|
||||
forked = self._read_forked_session_id()
|
||||
if not forked:
|
||||
return []
|
||||
sessions_root = self.logs_dir / "sessions" / "projects"
|
||||
if not sessions_root.is_dir():
|
||||
return []
|
||||
|
||||
all_jsonls: list[Path] = []
|
||||
for project_dir in sessions_root.iterdir():
|
||||
if not project_dir.is_dir():
|
||||
continue
|
||||
all_jsonls.extend(project_dir.glob("*.jsonl"))
|
||||
|
||||
has_forked_match = any(j.stem == forked for j in all_jsonls)
|
||||
if not has_forked_match:
|
||||
self.logger.warning(
|
||||
"Forked session id %s from claude-code.txt has no matching "
|
||||
"JSONL in %s (found: %s). Leaving sessions/ untouched so "
|
||||
"harbor's converter can attempt against what's there.",
|
||||
forked,
|
||||
sessions_root,
|
||||
[j.name for j in all_jsonls],
|
||||
)
|
||||
return []
|
||||
|
||||
moved: list[tuple[Path, Path]] = []
|
||||
for jsonl in all_jsonls:
|
||||
if jsonl.stem == forked:
|
||||
continue
|
||||
disabled = jsonl.with_suffix(jsonl.suffix + _DISABLED_SUFFIX)
|
||||
try:
|
||||
shutil.move(str(jsonl), str(disabled))
|
||||
except OSError as exc:
|
||||
# All-or-nothing: a partial move would feed the converter a
|
||||
# MIXED set (forked + still-present seeded), which is the
|
||||
# exact original failure mode this override exists to prevent.
|
||||
# Roll back any successful moves and let harbor's converter
|
||||
# run on the unmodified directory — same outcome as today
|
||||
# (likely fails, our WARNING fires), no worse.
|
||||
self.logger.warning(
|
||||
"Could not move seeded JSONL %s aside: %s. Rolling back "
|
||||
"any prior moves to avoid feeding the converter a mixed "
|
||||
"set.",
|
||||
jsonl,
|
||||
exc,
|
||||
)
|
||||
self._restore_moved_jsonls(moved)
|
||||
return []
|
||||
moved.append((jsonl, disabled))
|
||||
_log.info(
|
||||
"Moved seeded JSONL %s aside so harbor converter only "
|
||||
"sees forked session %s",
|
||||
jsonl.name,
|
||||
forked,
|
||||
)
|
||||
return moved
|
||||
|
||||
def _restore_moved_jsonls(self, moved: list[tuple[Path, Path]]) -> None:
|
||||
for original, disabled in moved:
|
||||
try:
|
||||
shutil.move(str(disabled), str(original))
|
||||
except OSError as exc:
|
||||
self.logger.warning(
|
||||
"Could not restore seeded JSONL %s from %s: %s",
|
||||
original,
|
||||
disabled,
|
||||
exc,
|
||||
)
|
||||
|
||||
def _summarize_sessions_dir(self) -> str:
|
||||
sessions_root = self.logs_dir / "sessions" / "projects"
|
||||
if not sessions_root.is_dir():
|
||||
return "missing"
|
||||
parts: list[str] = []
|
||||
for project_dir in sorted(sessions_root.iterdir()):
|
||||
if not project_dir.is_dir():
|
||||
continue
|
||||
jsonls = sorted(p.name for p in project_dir.glob("*.jsonl"))
|
||||
parts.append(f"{project_dir.name}={jsonls}")
|
||||
return ", ".join(parts) if parts else "no JSONLs"
|
||||
|
||||
|
||||
# --- TRANSITIONAL: full Claude Code built-in toolset -------------------------
|
||||
# These restore Claude Code's full built-in toolset (Read/Edit/Write/Grep/Glob/
|
||||
# Task/...) for snapshot session.jsonls recorded in the OLD full-toolset format.
|
||||
# A different toolset is a different agent, so they keep the original
|
||||
# "claude-code" / "snapshot-claude-code" names. DELETE this mixin + both classes
|
||||
# once every snapshot session.jsonl is re-recorded in the reduced format.
|
||||
|
||||
|
||||
class _BrowserToolsetMixin:
|
||||
"""The canonical reduced toolset PLUS the ``Read`` built-in, for tasks that opt into a
|
||||
browser: a screenshot is only useful to an agent that can look at it, and ``Read`` is what
|
||||
turns a PNG on disk into an image the model actually sees.
|
||||
|
||||
This is a SEPARATE AGENT, not a flag, per the rule that the toolset is chosen by which class
|
||||
harbor runs and the class name records it — so a benchmark row can never silently compare an
|
||||
agent that could see against one that couldn't.
|
||||
|
||||
Two things to be clear-eyed about:
|
||||
* ``Read`` is not image-only. It also reads text files, PDFs and notebooks, so these tasks
|
||||
get back a file-reading built-in the reduced toolset deliberately removes. There is no
|
||||
narrower built-in; an image-only MCP tool was rejected because the canonical agent avoids
|
||||
MCP (see the async-MCP startup race note at the top of this file).
|
||||
* Tasks on this agent are not comparable with tasks on the canonical one. That is the point
|
||||
of the distinct name.
|
||||
"""
|
||||
|
||||
def build_cli_flags(self) -> str:
|
||||
flags = ClaudeCode.build_cli_flags(self)
|
||||
note = _toolset_note(with_browser=self._has_browser, with_read=True)
|
||||
extra = f"--tools Bash,Read --append-system-prompt {shlex.quote(note)}"
|
||||
return f"{flags} {extra}" if flags else extra
|
||||
|
||||
|
||||
class BrowserPreinstalledClaudeCode(_BrowserToolsetMixin, PreinstalledClaudeCode):
|
||||
"""Reduced toolset + Read, manual (non-snapshot) tasks."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "claude-code-reduced-toolset-browser"
|
||||
|
||||
|
||||
class BrowserSnapshotClaudeCode(_BrowserToolsetMixin, SnapshotClaudeCode):
|
||||
"""Reduced toolset + Read, snapshot tasks."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "snapshot-claude-code-reduced-toolset-browser"
|
||||
|
||||
|
||||
class _FullToolsetMixin:
|
||||
"""Override the canonical reduced toolset back to Claude Code's stock full
|
||||
built-in toolset: no str_replace_editor CLI to stage, and no --tools / note.
|
||||
Mixed in BEFORE the reduced base so its install/build_cli_flags win, while the
|
||||
snapshot resume/fork and the claude-binary probe are still inherited."""
|
||||
|
||||
async def install(self, environment) -> None:
|
||||
# Full toolset has no editor CLI to stage — just ensure the claude binary.
|
||||
await self._ensure_claude_binary(environment)
|
||||
|
||||
def build_cli_flags(self) -> str:
|
||||
# Stock Claude Code flags: the full built-in toolset, no --tools, no note.
|
||||
return ClaudeCode.build_cli_flags(self)
|
||||
|
||||
|
||||
class FullToolsetPreinstalledClaudeCode(_FullToolsetMixin, PreinstalledClaudeCode):
|
||||
"""TRANSITIONAL full-toolset manual agent. Delete once snapshots are reduced-format."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "claude-code"
|
||||
|
||||
|
||||
class FullToolsetSnapshotClaudeCode(_FullToolsetMixin, SnapshotClaudeCode):
|
||||
"""TRANSITIONAL full-toolset snapshot agent. Delete once snapshots are reduced-format."""
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "snapshot-claude-code"
|
||||
@@ -0,0 +1,295 @@
|
||||
/**
|
||||
* stage-atomic-rubric.ts — stage a task's atomic rubric into the grading
|
||||
* copies the rubric grader modes read, so
|
||||
* `HARBOR_GRADER_MODE=rubric-trinary scripts/harbor-regrade …` can run inside
|
||||
* this toolkit.
|
||||
*
|
||||
* Source of truth: the task's own atomic rubric —
|
||||
* `tests/atomic-rubric.yaml` (the current name), or `tests/rubrics.yaml` on a
|
||||
* task converted before the rename. A task carrying BOTH names with different
|
||||
* content is a hard error: silently preferring either file could stage a
|
||||
* rubric that is not the one just edited, and the grader would grade the
|
||||
* wrong criteria with no signal.
|
||||
*
|
||||
* Staged into `harbor-tasks/<slug>/tests/`:
|
||||
* - `rubric-criteria.md` — the criteria text shown to the grader: one
|
||||
* "### Criterion: <id>" section per criterion, guideline and elaboration
|
||||
* only. Category, severity, and dimensions are stripped, so the grader
|
||||
* stays severity-blind.
|
||||
* - `rubric-criteria.json` — {task, criteria: [{id, category, severity,
|
||||
* dimensions}]} for `render-rubric-grade.py` (criterion-id validation and
|
||||
* severity-weighted aggregation). The grader never sees this file.
|
||||
* - `render-rubric-grade.py` — synced from `task-shared/` when the task's
|
||||
* copy is missing or differs from the shared source.
|
||||
*
|
||||
* `tests/grader-context.md` is part of the task's own package — this script
|
||||
* checks that it exists and never writes it. Write it alongside the rubric;
|
||||
* the `/write-atomic-rubric` skill covers both files.
|
||||
*
|
||||
* Staged files are derived from the rubric. Re-run this script after every
|
||||
* rubric edit, and run `--restore` to remove the staged copies. This script
|
||||
* validates structure only (readable YAML, unique criterion ids, at most two
|
||||
* Crux criteria); the `/detector-rubric-coverage` and `/detector-rubric-form`
|
||||
* skills are the content review.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/stage-atomic-rubric.ts <task-slug>
|
||||
* npx tsx scripts/stage-atomic-rubric.ts <task-slug> --restore
|
||||
*/
|
||||
|
||||
import { createHash } from 'node:crypto';
|
||||
import { copyFileSync, existsSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { createRequire } from 'node:module';
|
||||
import { basename, dirname, isAbsolute, join, relative, resolve } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import pino from 'pino';
|
||||
import pinoPretty from 'pino-pretty';
|
||||
import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const TOOLKIT_ROOT = resolve(__dirname, '..');
|
||||
const SHARED_DIR = join(TOOLKIT_ROOT, 'task-shared');
|
||||
|
||||
const argv = yargs(hideBin(process.argv))
|
||||
.usage('Usage: $0 <task> [options]')
|
||||
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
|
||||
.option('restore', {
|
||||
type: 'boolean',
|
||||
default: false,
|
||||
describe: 'Remove the files a previous staging created',
|
||||
})
|
||||
.option('json', { type: 'boolean', default: false, describe: 'Structured JSON logs' })
|
||||
.demandCommand(1, 'Name the task to stage: npx tsx scripts/stage-atomic-rubric.ts <task-slug>')
|
||||
.strict()
|
||||
.help()
|
||||
.parseSync();
|
||||
|
||||
const log = pino(
|
||||
{ name: 'stage-atomic-rubric', level: 'info' },
|
||||
argv.json
|
||||
? process.stdout
|
||||
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
|
||||
);
|
||||
|
||||
/** The atomic-rubric filenames, current name first. */
|
||||
const RUBRIC_NAMES = ['atomic-rubric.yaml', 'rubrics.yaml'] as const;
|
||||
|
||||
/** Record of exactly what staging created, so --restore removes only that. */
|
||||
const STAGE_MANIFEST = '.rubric-staged.json';
|
||||
|
||||
/** The rubric shape this script needs. Validation is structural only — the
|
||||
* rubric detectors and the repo-side validator own the content rules. */
|
||||
interface RubricCriterion {
|
||||
readonly id: string;
|
||||
readonly category: string;
|
||||
readonly severity?: string | null;
|
||||
readonly guideline: string;
|
||||
readonly elaboration?: string | null;
|
||||
readonly dimensions?: readonly string[];
|
||||
}
|
||||
|
||||
interface RubricsDoc {
|
||||
readonly task: string;
|
||||
readonly criteria: readonly RubricCriterion[];
|
||||
}
|
||||
|
||||
function fail(message: string): never {
|
||||
log.error(message);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
/** harbor-tasks/<slug> from a slug or a path, mirroring check-task-infra.ts. */
|
||||
function resolveTaskDir(task: string): string {
|
||||
const candidate = isAbsolute(task) ? task : resolve(process.cwd(), task);
|
||||
if (existsSync(join(candidate, 'task.toml'))) return candidate;
|
||||
const bySlug = join(TOOLKIT_ROOT, 'harbor-tasks', task);
|
||||
if (existsSync(join(bySlug, 'task.toml'))) return bySlug;
|
||||
return fail(`No task found at ${task} or harbor-tasks/${task} (expected a task.toml inside).`);
|
||||
}
|
||||
|
||||
const sha256 = (p: string): string => createHash('sha256').update(readFileSync(p)).digest('hex');
|
||||
|
||||
/** The task's rubric file — current name first, pre-rename name honored, both
|
||||
* present with different content refused. */
|
||||
function resolveRubricPath(testsDir: string): string {
|
||||
const present = RUBRIC_NAMES.map((name) => join(testsDir, name)).filter((p) => existsSync(p));
|
||||
if (present.length === 0) {
|
||||
return fail(
|
||||
`No atomic rubric found: expected tests/atomic-rubric.yaml ` +
|
||||
`(or tests/rubrics.yaml on a task converted before the rename). ` +
|
||||
`Write it with the /write-atomic-rubric skill first.`
|
||||
);
|
||||
}
|
||||
if (present.length > 1 && new Set(present.map(sha256)).size > 1) {
|
||||
return fail(
|
||||
`Both tests/atomic-rubric.yaml and tests/rubrics.yaml exist with different content. ` +
|
||||
`Keep exactly one; tests/atomic-rubric.yaml is the current name.`
|
||||
);
|
||||
}
|
||||
return present[0];
|
||||
}
|
||||
|
||||
/** Structural gate: the properties the staged outputs are built from. */
|
||||
function toRubricsDoc(raw: unknown, sourceName: string): RubricsDoc {
|
||||
if (typeof raw !== 'object' || raw === null || Array.isArray(raw)) {
|
||||
return fail(`${sourceName} is not a YAML mapping with task and criteria keys.`);
|
||||
}
|
||||
const doc = raw as { task?: unknown; criteria?: unknown };
|
||||
if (typeof doc.task !== 'string' || doc.task.length === 0) {
|
||||
return fail(`${sourceName} is missing the top-level task key.`);
|
||||
}
|
||||
if (!Array.isArray(doc.criteria) || doc.criteria.length === 0) {
|
||||
return fail(`${sourceName} has no criteria list.`);
|
||||
}
|
||||
const seen = new Set<string>();
|
||||
let cruxCount = 0;
|
||||
for (const [index, entry] of doc.criteria.entries()) {
|
||||
const criterion = entry as Partial<RubricCriterion> | null;
|
||||
if (typeof criterion !== 'object' || criterion === null) {
|
||||
return fail(`${sourceName} criteria[${index}] is not a mapping.`);
|
||||
}
|
||||
if (typeof criterion.id !== 'string' || criterion.id.length === 0) {
|
||||
return fail(`${sourceName} criteria[${index}] has no id.`);
|
||||
}
|
||||
if (seen.has(criterion.id)) {
|
||||
return fail(`${sourceName} has a duplicate criterion id: ${criterion.id}.`);
|
||||
}
|
||||
seen.add(criterion.id);
|
||||
if (typeof criterion.guideline !== 'string' || criterion.guideline.trim().length === 0) {
|
||||
return fail(`${sourceName} criterion ${criterion.id} has no guideline.`);
|
||||
}
|
||||
if (typeof criterion.category !== 'string' || criterion.category.length === 0) {
|
||||
return fail(`${sourceName} criterion ${criterion.id} has no category.`);
|
||||
}
|
||||
if (criterion.severity === 'crux') cruxCount += 1;
|
||||
}
|
||||
if (cruxCount > 2) {
|
||||
return fail(`${sourceName} designates ${cruxCount} Crux criteria; the cap is two.`);
|
||||
}
|
||||
return doc as RubricsDoc;
|
||||
}
|
||||
|
||||
/** Grader-facing view: guideline and elaboration only, severity-blind. */
|
||||
function renderCriteriaMarkdown(doc: RubricsDoc): string {
|
||||
const sections = doc.criteria.map((criterion) => {
|
||||
const parts = [`### Criterion: ${criterion.id}`, criterion.guideline.trim()];
|
||||
if (criterion.elaboration?.trim()) parts.push(criterion.elaboration.trim());
|
||||
return parts.join('\n\n');
|
||||
});
|
||||
return sections.join('\n\n') + '\n';
|
||||
}
|
||||
|
||||
function stage(taskDir: string): void {
|
||||
const testsDir = join(taskDir, 'tests');
|
||||
if (!existsSync(testsDir)) {
|
||||
return fail(`${relative(TOOLKIT_ROOT, taskDir)} has no tests/ directory.`);
|
||||
}
|
||||
const rubricPath = resolveRubricPath(testsDir);
|
||||
const sourceName = `tests/${basename(rubricPath)}`;
|
||||
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = parseYaml(readFileSync(rubricPath, 'utf8'));
|
||||
} catch (error) {
|
||||
return fail(`${sourceName} is not readable YAML: ${(error as Error).message}`);
|
||||
}
|
||||
const doc = toRubricsDoc(parsed, sourceName);
|
||||
|
||||
const created: string[] = [];
|
||||
const writeStaged = (name: string, content: string, mode?: number): void => {
|
||||
writeFileSync(join(testsDir, name), content, mode ? { mode } : undefined);
|
||||
created.push(name);
|
||||
};
|
||||
|
||||
writeStaged('rubric-criteria.md', renderCriteriaMarkdown(doc));
|
||||
writeStaged(
|
||||
'rubric-criteria.json',
|
||||
JSON.stringify(
|
||||
{
|
||||
task: doc.task,
|
||||
// severity feeds render-rubric-grade.py's severity-weighted
|
||||
// aggregation; dimensions ride along for offline slicing. The grader
|
||||
// never sees this file — severity-blindness lives in
|
||||
// rubric-criteria.md.
|
||||
criteria: doc.criteria.map((criterion) => ({
|
||||
id: criterion.id,
|
||||
category: criterion.category,
|
||||
severity: criterion.severity ?? null,
|
||||
dimensions: criterion.dimensions ?? [],
|
||||
})),
|
||||
},
|
||||
null,
|
||||
2
|
||||
) + '\n'
|
||||
);
|
||||
|
||||
// The renderer is a shared asset. Sync it so the regrade runs the current
|
||||
// weights; scripts/harbor-regrade performs the same self-heal.
|
||||
const rendererSource = join(SHARED_DIR, 'render-rubric-grade.py');
|
||||
const rendererDest = join(testsDir, 'render-rubric-grade.py');
|
||||
if (!existsSync(rendererSource)) {
|
||||
return fail('task-shared/render-rubric-grade.py is missing from this toolkit.');
|
||||
}
|
||||
if (!existsSync(rendererDest) || sha256(rendererDest) !== sha256(rendererSource)) {
|
||||
copyFileSync(rendererSource, rendererDest);
|
||||
created.push('render-rubric-grade.py');
|
||||
}
|
||||
|
||||
writeFileSync(join(testsDir, STAGE_MANIFEST), JSON.stringify({ created }, null, 2) + '\n');
|
||||
|
||||
if (!existsSync(join(testsDir, 'grader-context.md'))) {
|
||||
log.warn(
|
||||
'tests/grader-context.md is missing. The rubric grader modes read it beside the ' +
|
||||
'criteria; write it before running a rubric-mode regrade or packaging the task.'
|
||||
);
|
||||
}
|
||||
log.info(
|
||||
{ source: sourceName, criteria: doc.criteria.length, staged: created },
|
||||
'Staged the atomic-rubric grading copies.'
|
||||
);
|
||||
}
|
||||
|
||||
function restore(taskDir: string): void {
|
||||
const testsDir = join(taskDir, 'tests');
|
||||
const manifestPath = join(testsDir, STAGE_MANIFEST);
|
||||
if (!existsSync(manifestPath)) {
|
||||
log.warn('No staging manifest found; nothing to restore.');
|
||||
return;
|
||||
}
|
||||
let names: string[] = [];
|
||||
try {
|
||||
const manifest = JSON.parse(readFileSync(manifestPath, 'utf8')) as { created?: unknown };
|
||||
if (Array.isArray(manifest.created)) {
|
||||
names = manifest.created.filter((n): n is string => typeof n === 'string');
|
||||
}
|
||||
} catch {
|
||||
return fail(`${STAGE_MANIFEST} is unreadable; remove the staged files by hand.`);
|
||||
}
|
||||
for (const name of names) {
|
||||
// Only ever files this script wrote into tests/ — refuse anything else.
|
||||
if (name.includes('/') || name.includes('..')) continue;
|
||||
const filePath = join(testsDir, name);
|
||||
if (existsSync(filePath)) rmSync(filePath);
|
||||
}
|
||||
rmSync(manifestPath);
|
||||
log.info({ removed: names }, 'Removed the staged grading copies.');
|
||||
}
|
||||
|
||||
// The yaml package reaches containers created after it joined package.json;
|
||||
// a container created earlier has every other dependency but not this one,
|
||||
// so resolve it at run time and say what to do instead of crashing.
|
||||
let parseYaml: (src: string) => unknown;
|
||||
try {
|
||||
const requireFromHere = createRequire(fileURLToPath(import.meta.url));
|
||||
({ parse: parseYaml } = requireFromHere('yaml') as { parse: (src: string) => unknown });
|
||||
} catch {
|
||||
fail(
|
||||
'The yaml package is not installed in this container. Rebuild the Authoring ' +
|
||||
'container ("Dev Containers: Rebuild Container"), or run npm install in the toolkit root.'
|
||||
);
|
||||
}
|
||||
|
||||
const taskDir = resolveTaskDir(String(argv._[0]));
|
||||
if (argv.restore) restore(taskDir);
|
||||
else stage(taskDir);
|
||||
@@ -0,0 +1,146 @@
|
||||
/**
|
||||
* stamp-trial-inputs.ts — record, at trial LAUNCH time, the checksums of the
|
||||
* task inputs a harbor run is about to execute against, and stamp them into
|
||||
* the trial directories the run produces.
|
||||
*
|
||||
* Why launch time: copy-reference-run.ts used to capture checksums at COPY
|
||||
* time, which misses the headline staleness ordering — run trials, edit the
|
||||
* prompt, then copy the runs — and records the post-edit hashes (a genuinely
|
||||
* stale run then reads `fresh`). Harbor creates trial dirs itself (and, on
|
||||
* the daytona backend, populates them only at download after the trial), so
|
||||
* the earliest host-side point to capture is the moment `scripts/harbor-run`
|
||||
* launches: `capture` snapshots the inputs to a temp file before harbor
|
||||
* starts, and `apply` copies that snapshot into each trial dir once the job
|
||||
* directory exists. copy-reference-run.ts then prefers this run-time record
|
||||
* over its own capture-at-copy fallback.
|
||||
*
|
||||
* Usage (normally invoked by scripts/harbor-run, not by hand):
|
||||
* npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>
|
||||
* npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]
|
||||
*
|
||||
* Ships in the worker toolkit (see raccoon-worker-toolkit/package-worker-toolkit.ts),
|
||||
* so it must only import from its shipped file set — same constraint as
|
||||
* copy-reference-run.ts.
|
||||
*/
|
||||
|
||||
import './lib/check-devcontainer';
|
||||
|
||||
import { existsSync, readFileSync, readdirSync, statSync, writeFileSync } from 'fs';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { basename, join, resolve } from 'path';
|
||||
|
||||
import {
|
||||
INPUT_CHECKSUMS_FILENAME,
|
||||
captureTaskInputs,
|
||||
readTaskInputChecksums,
|
||||
} from './lib/input-checksums';
|
||||
|
||||
function usage(): never {
|
||||
console.error(
|
||||
[
|
||||
'Usage:',
|
||||
' npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>',
|
||||
' npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]',
|
||||
].join('\n')
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
/** Snapshot the task inputs as they stand at launch; write the record to `outFile`. */
|
||||
export function captureCommand(taskDir: string, outFile: string): void {
|
||||
if (!existsSync(taskDir)) {
|
||||
console.error(`Error: task dir ${taskDir} does not exist`);
|
||||
process.exit(1);
|
||||
}
|
||||
// taskSlug scopes `apply` to this task's trial dirs (concurrent harbor-runs
|
||||
// share harbor-jobs/) and makes any mis-routed stamp diagnosable later.
|
||||
const record = { ...captureTaskInputs(taskDir, 'run'), taskSlug: basename(resolve(taskDir)) };
|
||||
writeFileSync(outFile, JSON.stringify(record, null, 2) + '\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Does this trial dir belong to the task the capture was taken from? Two
|
||||
* signals, strongest first:
|
||||
*
|
||||
* 1. result.json `task_name` — written per-trial by harbor with the FULL,
|
||||
* unambiguous slug (org-prefixed for hub-published tasks). Authoritative
|
||||
* when readable, exactly as copy-reference-run.ts resolves trials.
|
||||
* 2. The trial DIRNAME's `<prefix>__<trialId>` prefix — harbor TRUNCATES
|
||||
* long slugs here, so the test is "the prefix is a truncation of the
|
||||
* slug", not equality. (Two tasks sharing a truncated prefix are told
|
||||
* apart by signal 1; the dirname alone can't distinguish them.)
|
||||
*/
|
||||
function trialBelongsToTask(trialDir: string, entry: string, slug: string): boolean {
|
||||
const sep = entry.lastIndexOf('__');
|
||||
if (sep === -1) return false;
|
||||
const resultPath = join(trialDir, 'result.json');
|
||||
if (existsSync(resultPath)) {
|
||||
try {
|
||||
const taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown })
|
||||
.task_name;
|
||||
if (typeof taskName === 'string' && taskName.length > 0) {
|
||||
return taskName.replace(/^[^/]+\//, '') === slug;
|
||||
}
|
||||
} catch {
|
||||
// Unparseable result.json — fall through to the dirname prefix.
|
||||
}
|
||||
}
|
||||
const prefix = entry.substring(0, sep);
|
||||
return prefix === slug || slug.startsWith(prefix);
|
||||
}
|
||||
|
||||
/**
|
||||
* Copy a launch-time capture into the capture's OWN task's trial dirs under
|
||||
* the given harbor job dir(s). Trial dirs are the `<slug>__<trialId>`
|
||||
* subdirectories harbor creates; anything else (stray files, harbor's own
|
||||
* metadata) is skipped — and so is any trial belonging to a DIFFERENT task:
|
||||
* several harbor-run invocations can share a cwd, and stamping another
|
||||
* task's trials with this capture's hashes would fabricate `capturedBy:
|
||||
* 'run'` evidence for inputs that task never ran against. An existing record
|
||||
* is left alone — it can only be from an earlier stamp of the same trial.
|
||||
*/
|
||||
export function applyCommand(captureFile: string, jobDirs: string[]): number {
|
||||
const record = readTaskInputChecksums(captureFile);
|
||||
if (!record || typeof record.taskSlug !== 'string' || record.taskSlug.length === 0) {
|
||||
console.error(
|
||||
`Error: ${captureFile} is not a readable input-checksums capture with a taskSlug`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
const slug = record.taskSlug;
|
||||
const raw = readFileSync(captureFile, 'utf-8');
|
||||
let stamped = 0;
|
||||
for (const jobDir of jobDirs) {
|
||||
if (!existsSync(jobDir)) continue;
|
||||
for (const entry of readdirSync(jobDir)) {
|
||||
const trialDir = join(jobDir, entry);
|
||||
if (!entry.includes('__') || !statSync(trialDir).isDirectory()) continue;
|
||||
if (!trialBelongsToTask(trialDir, entry, slug)) continue;
|
||||
const dest = join(trialDir, INPUT_CHECKSUMS_FILENAME);
|
||||
if (existsSync(dest)) continue;
|
||||
writeFileSync(dest, raw);
|
||||
stamped++;
|
||||
console.log(`Stamped ${dest}`);
|
||||
}
|
||||
}
|
||||
return stamped;
|
||||
}
|
||||
|
||||
// Main. Guarded so the test file can import the commands without running them.
|
||||
if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1])) {
|
||||
const [command, ...rest] = process.argv.slice(2);
|
||||
if (command === 'capture') {
|
||||
const outIdx = rest.indexOf('--out');
|
||||
const taskDir = rest.filter((a, i) => i !== outIdx && i !== outIdx + 1)[0];
|
||||
const outFile = outIdx !== -1 ? rest[outIdx + 1] : undefined;
|
||||
if (!taskDir || !outFile) usage();
|
||||
captureCommand(taskDir, outFile);
|
||||
} else if (command === 'apply') {
|
||||
const [captureFile, ...jobDirs] = rest;
|
||||
if (!captureFile || jobDirs.length === 0) usage();
|
||||
const stamped = applyCommand(captureFile, jobDirs);
|
||||
console.log(`Stamped ${stamped} trial dir(s) with launch-time input checksums`);
|
||||
} else {
|
||||
usage();
|
||||
}
|
||||
}
|
||||
93
worker-toolkit-potion-polyglot-orig/scripts/str_replace_editor
Executable file
93
worker-toolkit-potion-polyglot-orig/scripts/str_replace_editor
Executable file
@@ -0,0 +1,93 @@
|
||||
#!/usr/bin/env python3
|
||||
"""str_replace_editor — CLI-as-MCP wrapper around the vendored EditTool.
|
||||
|
||||
This is the "CLI-as-MCP" delivery of the `str_replace_editor` tool: the agent
|
||||
(which has ONLY the bash tool) invokes this script and passes the tool's
|
||||
arguments as one JSON object on stdin. The actual editing logic is the vendored
|
||||
`EditTool` under str_replace_editor_vendor/ (see VENDORED.md) — we add no
|
||||
behavior, we only:
|
||||
* instantiate it with run_command_preexec_fn=None (the class's own documented
|
||||
way to skip its uid/gid-1000 demotion, which would break writes in our
|
||||
sandbox where the workspace is owned by the agent user); and
|
||||
* adapt structured stdin-JSON <-> a bash-invokable CLI.
|
||||
|
||||
stdin: one JSON object, e.g.
|
||||
{"command":"view","path":"/workspace/app/models/x.rb"}
|
||||
{"command":"view","path":"/workspace/x.rb","view_range":[1,40]}
|
||||
{"command":"str_replace","path":"/workspace/x.rb","old_str":"a","new_str":"b"}
|
||||
{"command":"create","path":"/workspace/new.rb","file_text":"..."}
|
||||
{"command":"insert","path":"/workspace/x.rb","insert_line":10,"insert_text":"..."}
|
||||
stdout: the tool's result text (exit 0). stderr + exit 1: a tool error message.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from str_replace_editor_vendor.base import ToolError # noqa: E402
|
||||
from str_replace_editor_vendor.edit import EditTool # noqa: E402
|
||||
|
||||
# The keyword-only params the vendored EditTool.__call__ accepts.
|
||||
_ACCEPTED = {
|
||||
"command", "path", "file_text", "view_range",
|
||||
"old_str", "new_str", "insert_text", "insert_line",
|
||||
}
|
||||
|
||||
|
||||
async def _run(payload: dict):
|
||||
# Reject unknown keys instead of silently dropping them: a typo like
|
||||
# `old_string` (vs `old_str`) should be a clear argument error, not a
|
||||
# confusing failure deeper inside EditTool with the param silently missing.
|
||||
unknown = set(payload) - _ACCEPTED
|
||||
if unknown:
|
||||
raise ToolError(
|
||||
f"unknown argument(s): {', '.join(sorted(unknown))}. "
|
||||
f"accepted keys: {', '.join(sorted(_ACCEPTED))}."
|
||||
)
|
||||
kwargs = dict(payload)
|
||||
if "command" not in kwargs or "path" not in kwargs:
|
||||
raise ToolError("Both `command` and `path` are required.")
|
||||
# run_command_preexec_fn=None → no uid/gid demotion (see module docstring).
|
||||
tool = EditTool(run_command_preexec_fn=None)
|
||||
return await tool(**kwargs)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
raw = sys.stdin.read()
|
||||
if not raw.strip():
|
||||
sys.stderr.write("str_replace_editor: expected a JSON object on stdin\n")
|
||||
return 2
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
except json.JSONDecodeError as e:
|
||||
sys.stderr.write(f"str_replace_editor: invalid JSON on stdin: {e}\n")
|
||||
return 2
|
||||
if not isinstance(payload, dict):
|
||||
sys.stderr.write("str_replace_editor: stdin JSON must be an object\n")
|
||||
return 2
|
||||
try:
|
||||
result = asyncio.run(_run(payload))
|
||||
except ToolError as e:
|
||||
sys.stderr.write((e.message or "tool error") + "\n")
|
||||
return 1
|
||||
except TypeError as e:
|
||||
# e.g. an unexpected/duplicate kwarg shape — surface like a tool error.
|
||||
sys.stderr.write(f"str_replace_editor: bad arguments: {e}\n")
|
||||
return 1
|
||||
# EditTool returns a (CLI)Result with .output / .error / .base64_image / .system
|
||||
if getattr(result, "error", None):
|
||||
sys.stderr.write(result.error if result.error.endswith("\n") else result.error + "\n")
|
||||
if getattr(result, "system", None):
|
||||
sys.stderr.write(f"[system] {result.system}\n")
|
||||
out = getattr(result, "output", None) or ""
|
||||
if getattr(result, "base64_image", None):
|
||||
out += "\n(image content omitted in CLI mode)"
|
||||
if out:
|
||||
sys.stdout.write(out if out.endswith("\n") else out + "\n")
|
||||
return 1 if getattr(result, "error", None) else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1 @@
|
||||
"""Vendored verbatim — do not edit. See VENDORED.md for provenance."""
|
||||
@@ -0,0 +1,49 @@
|
||||
from dataclasses import dataclass, fields, replace
|
||||
|
||||
|
||||
@dataclass(kw_only=True, frozen=True)
|
||||
class ToolResult:
|
||||
"""Represents the result of a tool execution."""
|
||||
|
||||
output: str | None = None
|
||||
error: str | None = None
|
||||
base64_image: str | None = None
|
||||
system: str | None = None
|
||||
|
||||
def __bool__(self):
|
||||
return any(getattr(self, field.name) for field in fields(self))
|
||||
|
||||
def __add__(self, other: "ToolResult"):
|
||||
def combine_fields(field: str | None, other_field: str | None, concatenate: bool = True):
|
||||
if field and other_field:
|
||||
if concatenate:
|
||||
return field + other_field
|
||||
raise ValueError("Cannot combine tool results")
|
||||
return field or other_field
|
||||
|
||||
return ToolResult(
|
||||
output=combine_fields(self.output, other.output),
|
||||
error=combine_fields(self.error, other.error),
|
||||
base64_image=combine_fields(self.base64_image, other.base64_image, False),
|
||||
system=combine_fields(self.system, other.system),
|
||||
)
|
||||
|
||||
def replace(self, **kwargs):
|
||||
"""Returns a new ToolResult with the given fields replaced."""
|
||||
return replace(self, **kwargs)
|
||||
|
||||
|
||||
# QUESTION(simon): What's our intent behind differentiating here?
|
||||
class CLIResult(ToolResult):
|
||||
"""A ToolResult that can be rendered as a CLI output."""
|
||||
|
||||
|
||||
class ToolFailure(ToolResult):
|
||||
"""A ToolResult that represents a failure."""
|
||||
|
||||
|
||||
class ToolError(Exception):
|
||||
"""Raised when a tool encounters an error."""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
@@ -0,0 +1,476 @@
|
||||
import asyncio
|
||||
import base64
|
||||
import shlex
|
||||
from collections import deque
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Literal, get_args
|
||||
|
||||
from .base import CLIResult, ToolError, ToolResult
|
||||
from .run import demote, maybe_truncate, run
|
||||
|
||||
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
|
||||
|
||||
Command = Literal[
|
||||
"view",
|
||||
"create",
|
||||
"str_replace",
|
||||
"insert",
|
||||
]
|
||||
SNIPPET_LINES: int = 4
|
||||
|
||||
MAX_RESPONSE_LEN: int = 16000
|
||||
|
||||
|
||||
class EditTool:
|
||||
"""
|
||||
An filesystem editor tool that allows the agent to view, create, and edit files.
|
||||
The tool parameters are defined by Anthropic and are not editable.
|
||||
"""
|
||||
|
||||
def __init__(self, run_command_preexec_fn=demote):
|
||||
"""
|
||||
Initialize the EditTool.
|
||||
|
||||
Args:
|
||||
run_command_preexec_fn: Function to run in child process before executing
|
||||
shell commands via the run() utility.
|
||||
Defaults to demote() which drops privileges to uid/gid 1000.
|
||||
Pass None to skip preexec, or any callable for custom behavior.
|
||||
"""
|
||||
self._run_command_preexec_fn = run_command_preexec_fn
|
||||
|
||||
async def __call__(
|
||||
self,
|
||||
*,
|
||||
command: Command,
|
||||
path: str,
|
||||
file_text: str | None = None,
|
||||
view_range: list[int] | None = None,
|
||||
old_str: str | None = None,
|
||||
new_str: str | None = None,
|
||||
insert_text: str | None = None,
|
||||
insert_line: int | None = None,
|
||||
):
|
||||
_path = Path(path)
|
||||
self.validate_path(command, _path)
|
||||
if command == "view":
|
||||
return await self.view(_path, view_range)
|
||||
elif command == "create":
|
||||
if file_text is None:
|
||||
raise ToolError("Parameter `file_text` is required for command: create")
|
||||
await self.write_file(_path, file_text)
|
||||
return ToolResult(output=f"File created successfully at: {_path}")
|
||||
elif command == "str_replace":
|
||||
if old_str is None:
|
||||
raise ToolError("Parameter `old_str` is required for command: str_replace")
|
||||
return await self.str_replace(_path, old_str, new_str)
|
||||
elif command == "insert":
|
||||
if insert_line is None:
|
||||
raise ToolError("Parameter `insert_line` is required for command: insert")
|
||||
if insert_text is None:
|
||||
raise ToolError("Parameter `insert_text` is required for command: insert")
|
||||
return await self.insert(_path, insert_line, insert_text)
|
||||
raise ToolError(
|
||||
f"Unrecognized command {command}. The allowed commands for the {self.name} tool are: {', '.join(get_args(Command))}"
|
||||
)
|
||||
|
||||
def validate_path(self, command: str, path: Path):
|
||||
"""
|
||||
Check that the path/command combination is valid.
|
||||
"""
|
||||
# Check if its an absolute path
|
||||
if not path.is_absolute():
|
||||
suggested_path = Path("") / path
|
||||
raise ToolError(
|
||||
f"The path {path} is not an absolute path, it should start with `/`. Maybe you meant {suggested_path}?"
|
||||
)
|
||||
# Check if path exists
|
||||
if not path.exists() and command != "create":
|
||||
raise ToolError(f"The path {path} does not exist. Please provide a valid path.")
|
||||
if path.exists() and command == "create":
|
||||
raise ToolError(f"File already exists at: {path}. Cannot overwrite files using command `create`.")
|
||||
# Check if the path points to a directory
|
||||
if path.is_dir():
|
||||
if command != "view":
|
||||
raise ToolError(
|
||||
f"The path {path} is a directory and only the `view` command can be used on directories"
|
||||
)
|
||||
|
||||
async def view(self, path: Path, view_range: list[int] | None = None):
|
||||
"""Implement the view command"""
|
||||
if path.is_dir():
|
||||
if view_range:
|
||||
raise ToolError("The `view_range` parameter is not allowed when `path` points to a directory.")
|
||||
|
||||
_, stdout, stderr = await run(
|
||||
rf"find {path} -maxdepth 2 -not -path '*/\.*'", preexec_fn=self._run_command_preexec_fn
|
||||
)
|
||||
if not stderr:
|
||||
stdout = f"Here's the files and directories up to 2 levels deep in {path}, excluding hidden items:\n{stdout}\n"
|
||||
return CLIResult(output=stdout, error=stderr)
|
||||
|
||||
image_extensions = {'.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tiff', '.tif', '.webp', '.svg', '.ico'}
|
||||
if path.suffix.lower() in image_extensions:
|
||||
if view_range:
|
||||
raise ToolError("The `view_range` parameter is not allowed when `path` points to an image file.")
|
||||
|
||||
try:
|
||||
image_bytes = path.read_bytes()
|
||||
base64_encoded = base64.b64encode(image_bytes).decode()
|
||||
|
||||
return CLIResult(
|
||||
output=f"Displaying image file: {path}",
|
||||
base64_image=base64_encoded
|
||||
)
|
||||
except Exception as e:
|
||||
raise ToolError(f"Failed to read image file {path}: {e}") from None
|
||||
|
||||
file_content = await self.read_file(path, truncate_after=None)
|
||||
file_text_lines = file_content.splitlines(keepends=True)
|
||||
n_lines_file = len(file_text_lines) + (1 if file_content.endswith(("\n", "\r\n", "\r")) else 0)
|
||||
|
||||
if view_range:
|
||||
if len(view_range) != 2 or not all(isinstance(i, int) for i in view_range):
|
||||
raise ToolError("Invalid `view_range`. It should be a list of two integers.")
|
||||
init_line, final_line = view_range
|
||||
if init_line < 1 or init_line > n_lines_file:
|
||||
raise ToolError(
|
||||
f"Invalid `view_range`: {view_range}. Its first element `{init_line}` should be within the range of lines of the file: {[1, n_lines_file]}"
|
||||
)
|
||||
if final_line > n_lines_file:
|
||||
raise ToolError(
|
||||
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be smaller than the number of lines in the file: `{n_lines_file}`"
|
||||
)
|
||||
if final_line != -1 and final_line < init_line:
|
||||
raise ToolError(
|
||||
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be larger or equal than its first `{init_line}`"
|
||||
)
|
||||
|
||||
# Extract only the requested lines
|
||||
if final_line != -1:
|
||||
selected_lines = file_text_lines[max(view_range[0] - 1, 0) : view_range[1]]
|
||||
else:
|
||||
selected_lines = file_text_lines[max(view_range[0] - 1, 0) :]
|
||||
# Join without modifying the original line endings
|
||||
file_content = "".join(selected_lines)
|
||||
|
||||
file_content = process_view_output_str(
|
||||
file_text=file_content,
|
||||
path=str(path),
|
||||
total_path_lines=n_lines_file,
|
||||
max_resp_ln=MAX_RESPONSE_LEN,
|
||||
view_range=(view_range[0], view_range[1]) if view_range else None,
|
||||
)
|
||||
|
||||
return CLIResult(output=file_content)
|
||||
|
||||
async def str_replace(self, path: Path, old_str: str, new_str: str | None):
|
||||
"""Implement the str_replace command, which replaces old_str with new_str in the file content"""
|
||||
# Read the file content
|
||||
file_content = await self.read_file(path, truncate_after=None)
|
||||
new_str = new_str if new_str is not None else ""
|
||||
|
||||
# Check if old_str is unique in the file
|
||||
occurrences = file_content.count(old_str)
|
||||
if occurrences == 0:
|
||||
raise ToolError(f"No replacement was performed, old_str `{old_str}` did not appear verbatim in {path}.")
|
||||
elif occurrences > 1:
|
||||
file_content_lines = file_content.split("\n")
|
||||
lines = [idx + 1 for idx, line in enumerate(file_content_lines) if old_str in line]
|
||||
raise ToolError(
|
||||
f"No replacement was performed. Multiple occurrences of old_str `{old_str}` in lines {lines}. Please ensure it is unique"
|
||||
)
|
||||
|
||||
# Replace old_str with new_str
|
||||
new_file_content = file_content.replace(old_str, new_str)
|
||||
|
||||
# Write the new content to the file
|
||||
await self.write_file(path, new_file_content)
|
||||
|
||||
# Create a snippet of the edited section
|
||||
replacement_line = file_content.split(old_str)[0].count("\n")
|
||||
start_line = max(0, replacement_line - SNIPPET_LINES)
|
||||
end_line = replacement_line + SNIPPET_LINES + new_str.count("\n")
|
||||
snippet = "\n".join(new_file_content.split("\n")[start_line : end_line + 1])
|
||||
|
||||
# Prepare the success message
|
||||
success_msg = f"The file {path} has been edited. "
|
||||
success_msg += self._make_output(snippet, f"a snippet of {path}", start_line + 1)
|
||||
success_msg += "Review the changes and make sure they are as expected. Edit the file again if necessary."
|
||||
|
||||
return CLIResult(output=success_msg)
|
||||
|
||||
async def insert(self, path: Path, insert_line: int, new_str: str):
|
||||
"""Implement the insert command, which inserts new_str at the specified line in the file content."""
|
||||
file_text = await self.read_file(path, truncate_after=None)
|
||||
file_text_lines = file_text.split("\n")
|
||||
n_lines_file = len(file_text_lines)
|
||||
|
||||
if insert_line < 0 or insert_line > n_lines_file:
|
||||
raise ToolError(
|
||||
f"Invalid `insert_line` parameter: {insert_line}. It should be within the range of lines of the file: {[0, n_lines_file]}"
|
||||
)
|
||||
|
||||
new_str_lines = new_str.split("\n")
|
||||
new_file_text_lines = file_text_lines[:insert_line] + new_str_lines + file_text_lines[insert_line:]
|
||||
snippet_lines = (
|
||||
file_text_lines[max(0, insert_line - SNIPPET_LINES) : insert_line]
|
||||
+ new_str_lines
|
||||
+ file_text_lines[insert_line : insert_line + SNIPPET_LINES]
|
||||
)
|
||||
|
||||
new_file_text = "\n".join(new_file_text_lines)
|
||||
snippet = "\n".join(snippet_lines)
|
||||
|
||||
await self.write_file(path, new_file_text)
|
||||
|
||||
success_msg = f"The file {path} has been edited. "
|
||||
success_msg += self._make_output(
|
||||
snippet,
|
||||
"a snippet of the edited file",
|
||||
max(1, insert_line - SNIPPET_LINES + 1),
|
||||
)
|
||||
success_msg += "Review the changes and make sure they are as expected (correct indentation, no duplicate lines, etc). Edit the file again if necessary."
|
||||
return CLIResult(output=success_msg)
|
||||
|
||||
async def read_file(self, path: Path, truncate_after: int | None = MAX_RESPONSE_LEN):
|
||||
"""Read the content of a file from a given path; raise a ToolError if an error occurs."""
|
||||
try:
|
||||
code, out, err = await run(
|
||||
f"cat {shlex.quote(str(path))}", truncate_after=truncate_after, preexec_fn=self._run_command_preexec_fn
|
||||
)
|
||||
if code != 0:
|
||||
raise ToolError(f"Ran into {err} while trying to read {path}")
|
||||
return out
|
||||
except Exception as e:
|
||||
print(e)
|
||||
raise ToolError(f"Ran into {e} while trying to read {path}") from None
|
||||
|
||||
async def write_file(self, path: Path, file: str):
|
||||
"""Write the content of a file to a given path; raise a ToolError if an error occurs."""
|
||||
try:
|
||||
# Write using stdin to avoid argument size limits
|
||||
process = await asyncio.create_subprocess_shell(
|
||||
f"cat > {shlex.quote(str(path))}",
|
||||
stdin=asyncio.subprocess.PIPE,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
preexec_fn=self._run_command_preexec_fn,
|
||||
)
|
||||
|
||||
stdout, stderr = await asyncio.wait_for(
|
||||
process.communicate(input=file.encode('utf-8')),
|
||||
timeout=120.0
|
||||
)
|
||||
|
||||
if process.returncode != 0:
|
||||
raise ToolError(f"Ran into {stderr.decode()} while trying to write to {path}")
|
||||
except asyncio.TimeoutError:
|
||||
raise ToolError(f"Timed out while trying to write to {path}")
|
||||
except Exception as e:
|
||||
raise ToolError(f"Ran into {e} while trying to write to {path}") from None
|
||||
|
||||
def _make_output(
|
||||
self,
|
||||
file_content: str,
|
||||
file_descriptor: str,
|
||||
init_line: int = 1,
|
||||
expand_tabs: bool = True,
|
||||
):
|
||||
"""Generate output for the CLI based on the content of a file."""
|
||||
file_content = maybe_truncate(file_content)
|
||||
if expand_tabs:
|
||||
file_content = file_content.expandtabs()
|
||||
file_content = "\n".join([f"{i + init_line:6}\t{line}" for i, line in enumerate(file_content.split("\n"))])
|
||||
return f"Here's the result of running `cat -n` on {file_descriptor}:\n" + file_content + "\n"
|
||||
|
||||
|
||||
### AUX utilities
|
||||
|
||||
|
||||
def add_line_numbers(text: str, includes_final_line: bool, n_first_line: int = 1) -> str:
|
||||
"""
|
||||
Given a string, returns the string with line numbers prepended to each line.
|
||||
|
||||
This function:
|
||||
- Preserves the original line endings (CR, LF, or CRLF) of each line
|
||||
- Adds a tab-separated line number prefix to each line
|
||||
- If the text ends with any newline character (\n, \r\n, or \r), adds an
|
||||
additional empty numbered line to represent the terminal empty line
|
||||
"""
|
||||
lines_with_endings = text.splitlines(keepends=True)
|
||||
result = [f"{ind + n_first_line:6}\t{line_with_ending}" for ind, line_with_ending in enumerate(lines_with_endings)]
|
||||
|
||||
# Add an extra empty line with line number if original text ends with newline
|
||||
if includes_final_line and text.endswith(("\n", "\r\n", "\r")):
|
||||
result.append(f"{len(lines_with_endings) + n_first_line:6}\t")
|
||||
|
||||
return "".join(result)
|
||||
|
||||
|
||||
def process_view_output_str(
|
||||
file_text: str,
|
||||
path: str,
|
||||
total_path_lines: int,
|
||||
max_resp_ln: int,
|
||||
view_range: tuple[int, int] | None = None,
|
||||
) -> str:
|
||||
# Get header
|
||||
header = f"Here's the content of {path} with line numbers"
|
||||
if total_path_lines is not None and view_range is not None:
|
||||
header += f" (which has a total of {total_path_lines} lines) with view_range={list(view_range)}"
|
||||
|
||||
# See if final line is included in the view_range
|
||||
if view_range is None or view_range[1] == -1 or view_range[1] == total_path_lines:
|
||||
includes_final_line = True
|
||||
else:
|
||||
includes_final_line = False
|
||||
n_first_line = view_range[0] if view_range is not None else 1
|
||||
|
||||
# Truncate if needed
|
||||
maybe_truncated_str = truncate_from_middle_v2(ss=file_text, max_len=max_resp_ln, n_line_offset=n_first_line - 1)
|
||||
if isinstance(maybe_truncated_str, str):
|
||||
# No truncation
|
||||
file_text_with_line_numbers = add_line_numbers(
|
||||
file_text,
|
||||
includes_final_line=includes_final_line,
|
||||
n_first_line=n_first_line,
|
||||
)
|
||||
else:
|
||||
# Truncation occurred
|
||||
before_with_line_numbers = add_line_numbers(
|
||||
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.before_lines)),
|
||||
includes_final_line=False,
|
||||
n_first_line=n_first_line,
|
||||
)
|
||||
if maybe_truncated_str.single_line:
|
||||
file_text_with_line_numbers = before_with_line_numbers
|
||||
else:
|
||||
after_with_line_numbers = add_line_numbers(
|
||||
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.after_lines)),
|
||||
includes_final_line=includes_final_line,
|
||||
n_first_line=1 + maybe_truncated_str.truncated_end_line,
|
||||
)
|
||||
file_text_with_line_numbers = (
|
||||
before_with_line_numbers + f"\t{maybe_truncated_str.truncation_msg}" + after_with_line_numbers
|
||||
)
|
||||
|
||||
# Add context-aware truncation message
|
||||
if view_range is not None:
|
||||
# User already using view_range, suggest adjusting it
|
||||
truncation_note = "\n<response clipped><NOTE>To save on context only part of the view range has been shown. You can adjust the view_range parameters or use `grep -n` to find specific content.</NOTE>"
|
||||
else:
|
||||
# User viewing whole file, suggest view_range or grep
|
||||
truncation_note = "\n<response clipped><NOTE>To save on context only part of this file has been shown to you. You can use view_range=[start_line, end_line] to see specific sections, or use `grep -n` to find what you're looking for.</NOTE>"
|
||||
|
||||
file_text_with_line_numbers += truncation_note
|
||||
|
||||
return f"{header}:\n{file_text_with_line_numbers}"
|
||||
|
||||
|
||||
@dataclass
|
||||
class TruncatedString:
|
||||
# Blocks
|
||||
before_lines: list[str]
|
||||
middle_lines: list[str]
|
||||
after_lines: list[str]
|
||||
|
||||
# Line numbers (starting from 1)
|
||||
truncated_start_line: int
|
||||
truncated_end_line: int
|
||||
|
||||
# Truncation msg
|
||||
truncation_msg: str
|
||||
single_line: bool
|
||||
|
||||
def as_str(self, lines: list[str]) -> str:
|
||||
return "".join(lines)
|
||||
|
||||
@property
|
||||
def full_truncated_str(self) -> str:
|
||||
return "".join(self.before_lines + [self.truncation_msg] + self.after_lines)
|
||||
|
||||
|
||||
def truncate_from_middle_v2(ss: str, max_len: int, n_line_offset: int = 0) -> "str | TruncatedString":
|
||||
"""
|
||||
If no truncation is needed, returns the original string.
|
||||
If truncation is needed, returns TruncatedString
|
||||
"""
|
||||
# No truncation needed
|
||||
if len(ss) <= max_len:
|
||||
return ss
|
||||
|
||||
# Single line
|
||||
lines_with_endings = ss.splitlines(True)
|
||||
if len(lines_with_endings) == 1:
|
||||
chars_per_side = max(1, max_len // 2)
|
||||
truncated_char_count = len(ss) - (chars_per_side * 2)
|
||||
truncation_msg = f"...< truncated {truncated_char_count} characters >..."
|
||||
|
||||
before_lines = [ss[:chars_per_side] + truncation_msg + ss[-chars_per_side:]]
|
||||
|
||||
return TruncatedString(
|
||||
before_lines=before_lines,
|
||||
middle_lines=[],
|
||||
after_lines=[],
|
||||
truncated_start_line=1 + n_line_offset,
|
||||
truncated_end_line=1 + n_line_offset,
|
||||
truncation_msg=truncation_msg,
|
||||
single_line=True,
|
||||
)
|
||||
|
||||
# Line truncation
|
||||
current_len = 0
|
||||
before_lines = []
|
||||
middle_lines = deque(lines_with_endings)
|
||||
after_lines = deque([])
|
||||
while current_len < max_len and len(middle_lines) > 1:
|
||||
# Before
|
||||
before_candidate_line = middle_lines[0]
|
||||
if len(before_candidate_line) + current_len <= max_len:
|
||||
before_lines.append(middle_lines.popleft())
|
||||
current_len += len(before_candidate_line)
|
||||
else:
|
||||
break
|
||||
|
||||
# After
|
||||
if len(middle_lines) > 1:
|
||||
after_candidate_line = middle_lines[-1]
|
||||
if len(after_candidate_line) + current_len <= max_len:
|
||||
after_lines.appendleft(middle_lines.pop())
|
||||
current_len += len(after_candidate_line)
|
||||
else:
|
||||
break
|
||||
|
||||
# Find truncated lines
|
||||
first_truncated_line = 1 + len(before_lines) + n_line_offset
|
||||
last_truncated_line = first_truncated_line + len(middle_lines) - 1
|
||||
if ss.endswith(("\n", "\r", "\r\n")) and len(after_lines) == 0:
|
||||
last_truncated_line += 1
|
||||
|
||||
# Create truncation msg
|
||||
if first_truncated_line == last_truncated_line:
|
||||
truncation_msg = f"< truncated line {first_truncated_line} >"
|
||||
else:
|
||||
truncation_msg = f"< truncated lines {first_truncated_line}-{last_truncated_line} >"
|
||||
if len(after_lines) != 0:
|
||||
if before_lines[0].endswith("\r\n"):
|
||||
truncation_msg += "\r\n"
|
||||
elif before_lines[0].endswith("\r"):
|
||||
truncation_msg += "\r"
|
||||
else:
|
||||
truncation_msg += "\n"
|
||||
|
||||
return TruncatedString(
|
||||
# Blocks
|
||||
before_lines=before_lines,
|
||||
middle_lines=list(middle_lines),
|
||||
after_lines=list(after_lines),
|
||||
# Line numbers (starting from 1)
|
||||
truncated_start_line=first_truncated_line,
|
||||
truncated_end_line=last_truncated_line,
|
||||
# Truncation msg
|
||||
truncation_msg=truncation_msg,
|
||||
single_line=False,
|
||||
)
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Utility to run shell commands asynchronously with a timeout."""
|
||||
|
||||
import asyncio # noqa -- swapping to trio would be beneficial, but not blocking atm
|
||||
import os
|
||||
|
||||
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
|
||||
MAX_RESPONSE_LEN: int = 16000
|
||||
|
||||
|
||||
def maybe_truncate(content: str, truncate_after: int | None = MAX_RESPONSE_LEN):
|
||||
"""Truncate content and append a notice if content exceeds the specified length."""
|
||||
return (
|
||||
content
|
||||
if not truncate_after or len(content) <= truncate_after
|
||||
else content[:truncate_after] + TRUNCATED_MESSAGE
|
||||
)
|
||||
|
||||
|
||||
def demote():
|
||||
"""Drop privileges to uid/gid 1000 for security.
|
||||
|
||||
This function is intended to be used as a preexec_fn in subprocess calls
|
||||
to ensure commands run with reduced privileges.
|
||||
"""
|
||||
os.setgid(1000)
|
||||
os.setuid(1000)
|
||||
|
||||
|
||||
async def run(
|
||||
cmd: str,
|
||||
timeout: float | None = 120.0, # seconds # noqa: ASYNC109
|
||||
truncate_after: int | None = MAX_RESPONSE_LEN,
|
||||
preexec_fn=demote,
|
||||
):
|
||||
"""Run a shell command asynchronously with a timeout.
|
||||
|
||||
Args:
|
||||
cmd: Command to execute
|
||||
timeout: Command timeout in seconds
|
||||
truncate_after: Maximum response length before truncation
|
||||
preexec_fn: Function to run in child process before exec (default: demote).
|
||||
Pass None to skip preexec, or any callable for custom behavior.
|
||||
|
||||
Returns:
|
||||
Tuple of (return_code, stdout, stderr)
|
||||
"""
|
||||
process = await asyncio.create_subprocess_shell(
|
||||
cmd,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
preexec_fn=preexec_fn,
|
||||
)
|
||||
|
||||
try:
|
||||
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout)
|
||||
return (
|
||||
process.returncode or 0,
|
||||
maybe_truncate(stdout.decode(), truncate_after=truncate_after),
|
||||
maybe_truncate(stderr.decode(), truncate_after=truncate_after),
|
||||
)
|
||||
except TimeoutError as exc:
|
||||
try:
|
||||
process.kill()
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
raise TimeoutError(f"Command '{cmd}' timed out after {timeout} seconds") from exc
|
||||
1162
worker-toolkit-potion-polyglot-orig/scripts/submit-task.ts
Normal file
1162
worker-toolkit-potion-polyglot-orig/scripts/submit-task.ts
Normal file
File diff suppressed because it is too large
Load Diff
20
worker-toolkit-potion-polyglot-orig/scripts/toolset_note.md
Normal file
20
worker-toolkit-potion-polyglot-orig/scripts/toolset_note.md
Normal file
@@ -0,0 +1,20 @@
|
||||
# Your actual toolset (this overrides any earlier tool guidance above)
|
||||
|
||||
This harness gives you exactly two ways to act, both through the `Bash` tool:
|
||||
|
||||
1. **Shell commands** for everything read-only and for running things: view and search files with `cat`, `sed -n`, `grep -rn`, `find`, `ls`; run tests; run `git`; etc.
|
||||
2. **A `str_replace_editor` file editor**, which you invoke from Bash by piping ONE JSON object on stdin to `/opt/agent-cli/str_replace_editor`. Use a quoted heredoc so backslashes and quotes survive:
|
||||
|
||||
`/opt/agent-cli/str_replace_editor <<'EDITOR'` then a line of JSON then `EDITOR`
|
||||
|
||||
The JSON `"command"` field selects the operation:
|
||||
- `view` — view a file (optionally `"view_range":[start,end]`) or list a directory: `{"command":"view","path":"/abs/file.rb"}`
|
||||
- `create` — create a NEW file (fails if it exists): `{"command":"create","path":"/abs/new.rb","file_text":"..."}`
|
||||
- `str_replace` — replace a UNIQUE substring: `{"command":"str_replace","path":"/abs/file.rb","old_str":"...","new_str":"..."}`
|
||||
- `insert` — insert text after a line: `{"command":"insert","path":"/abs/file.rb","insert_line":N,"insert_text":"..."}`
|
||||
|
||||
Paths must be absolute. Inside JSON strings, escape newlines as `\n` and double-quotes as `\"`.
|
||||
|
||||
There are **no** `Read`, `Grep`, `Glob`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite`, or `AskUserQuestion` tools — `Bash` is your only built-in tool. So disregard the earlier "Prefer the dedicated file/search tools over shell commands" guidance and the Memory section's "use the Write tool" instruction: those tools are not available in this harness. Search and read with shell commands; view, create, and edit files with `str_replace_editor`.
|
||||
|
||||
There is also no tool for asking the user an interactive question. If you need to ask the user something, or raise a concern about the request before acting on it, put it in your normal text response.
|
||||
@@ -0,0 +1,4 @@
|
||||
## Browser
|
||||
|
||||
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
|
||||
`require("playwright")` resolvable (CommonJS — `import` will not find it).
|
||||
@@ -0,0 +1,7 @@
|
||||
## Correction to the toolset above: you also have `Read`
|
||||
|
||||
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
|
||||
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
|
||||
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
|
||||
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
|
||||
edit files with `str_replace_editor`.
|
||||
88
worker-toolkit-potion-polyglot-orig/scripts/validate_task_dir.py
Executable file
88
worker-toolkit-potion-polyglot-orig/scripts/validate_task_dir.py
Executable file
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""validate_task_dir.py — say WHY harbor will not accept a task directory.
|
||||
|
||||
``harbor run -p <dir>`` silently reinterprets a directory that fails task
|
||||
validation as a *dataset* of tasks, finds none inside, and dies with
|
||||
``ValueError: Either datasets or tasks must be provided.`` — a message naming
|
||||
neither the path nor the missing file. ``scripts/harbor-run`` calls this first so
|
||||
the author reads "tests/test.sh is missing" instead.
|
||||
|
||||
Must run under HARBOR'S interpreter (its uv-tool venv), not any python3.11+: it
|
||||
imports harbor to reuse ``Task.is_valid_dir``, the exact predicate the CLI
|
||||
branches on, so the two cannot drift.
|
||||
|
||||
Usage: validate_task_dir.py <task-dir> [--disable-verification]
|
||||
Prints ``verdict=valid`` or ``verdict=invalid`` on stdout; the reason goes to
|
||||
stderr. Callers must gate on the stdout verdict, never on the exit code alone —
|
||||
an interpreter that cannot run this file at all also exits non-zero.
|
||||
Exit 0 = valid, 1 = invalid, 2 = the check could not run.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def reason(task_dir: Path, disable_verification: bool) -> str | None:
|
||||
"""Return why harbor rejects task_dir, or None if it accepts it."""
|
||||
from harbor.models.task.config import TaskConfig
|
||||
from harbor.models.task.paths import TaskPaths
|
||||
from harbor.models.task.task import Task
|
||||
|
||||
if Task.is_valid_dir(task_dir, disable_verification=disable_verification):
|
||||
return None
|
||||
|
||||
paths = TaskPaths(task_dir)
|
||||
if not paths.config_path.exists():
|
||||
return f"{paths.config_path} is missing."
|
||||
if not paths.environment_dir.exists():
|
||||
return f"{paths.environment_dir} is missing."
|
||||
try:
|
||||
config = TaskConfig.model_validate_toml(paths.config_path.read_text())
|
||||
except Exception as exc:
|
||||
return f"{paths.config_path} does not parse as a task config: {exc}"
|
||||
|
||||
# A stepped task carries no root instruction.md, so only the shape harbor
|
||||
# checks may be asserted here — hence steps first, root instruction last.
|
||||
if disable_verification:
|
||||
for step in config.steps or []:
|
||||
if not paths.step_dir(step.name).exists():
|
||||
return f"{paths.step_dir(step.name)} is missing."
|
||||
if not paths.step_instruction_path(step.name).exists():
|
||||
return f"{paths.step_instruction_path(step.name)} is missing."
|
||||
if not config.steps and not paths.instruction_path.exists():
|
||||
return f"{paths.instruction_path} is missing."
|
||||
else:
|
||||
# Private, but it owns the instruction/test diagnostics is_valid_dir discards.
|
||||
try:
|
||||
Task._validate_tests(config, paths)
|
||||
except FileNotFoundError as exc:
|
||||
return str(exc)
|
||||
except AttributeError:
|
||||
pass
|
||||
return f"{task_dir} is not a task directory harbor recognizes."
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = sys.argv[1:]
|
||||
disable_verification = "--disable-verification" in args
|
||||
positional = [a for a in args if not a.startswith("-")]
|
||||
if len(positional) != 1:
|
||||
print(
|
||||
f"usage: {sys.argv[0]} <task-dir> [--disable-verification]", file=sys.stderr
|
||||
)
|
||||
return 2
|
||||
try:
|
||||
why = reason(Path(positional[0]), disable_verification)
|
||||
except Exception as exc:
|
||||
print(f"validate_task_dir: check did not run ({exc})", file=sys.stderr)
|
||||
return 2
|
||||
if why is None:
|
||||
print("verdict=valid")
|
||||
return 0
|
||||
print("verdict=invalid")
|
||||
print(why, file=sys.stderr)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
41
worker-toolkit-potion-polyglot-orig/scripts/welcome.sh
Executable file
41
worker-toolkit-potion-polyglot-orig/scripts/welcome.sh
Executable file
@@ -0,0 +1,41 @@
|
||||
#!/bin/bash
|
||||
# Welcome banner for raccoon dev containers
|
||||
|
||||
CYAN='\033[1;36m'
|
||||
YELLOW='\033[1;33m'
|
||||
GRAY='\033[0;90m'
|
||||
RESET='\033[0m'
|
||||
|
||||
CONTAINER_TYPE="${1:-explore}"
|
||||
|
||||
if [ "$CONTAINER_TYPE" = "explore" ]; then
|
||||
COLOR="$CYAN"
|
||||
else
|
||||
COLOR="$YELLOW"
|
||||
fi
|
||||
|
||||
cat << 'RACCOON'
|
||||
|
||||
.----------------. .----------------. .----------------. .----------------. .----------------. .----------------. .-----------------.
|
||||
| .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. |
|
||||
| | _______ | || | __ | || | ______ | || | ______ | || | ____ | || | ____ | || | ____ _____ | |
|
||||
| | |_ __ \ | || | / \ | || | .' ___ | | || | .' ___ | | || | .' `. | || | .' `. | || ||_ \|_ _| | |
|
||||
| | | |__) | | || | / /\ \ | || | / .' \_| | || | / .' \_| | || | / .--. \ | || | / .--. \ | || | | \ | | | |
|
||||
| | | __ / | || | / ____ \ | || | | | | || | | | | || | | | | | | || | | | | | | || | | |\ \| | | |
|
||||
| | _| | \ \_ | || | _/ / \ \_ | || | \ `.___.'\ | || | \ `.___.'\ | || | \ `--' / | || | \ `--' / | || | _| |_\ |_ | |
|
||||
| | |____| |___| | || ||____| |____|| || | `._____.' | || | `._____.' | || | `.____.' | || | `.____.' | || ||_____|\____| | |
|
||||
| | | || | | || | | || | | || | | || | | || | | |
|
||||
| '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' |
|
||||
'----------------' '----------------' '----------------' '----------------' '----------------' '----------------' '----------------'
|
||||
|
||||
__ .-.
|
||||
.-"` .`'. /\\|
|
||||
_(\-/)_" , . ,\ /\\\/
|
||||
{(#b^d#)} . ./, |/\\\/
|
||||
`-.(Y).-` , | , |\.-`
|
||||
/~/,_/~~~\,__.-`
|
||||
////~ // ~\\
|
||||
==`==` ==` ==`
|
||||
------------------------------------------------
|
||||
|
||||
RACCOON
|
||||
Reference in New Issue
Block a user