added stocks app codebase and md

This commit is contained in:
2026-08-10 21:38:03 -04:00
parent 87f070f033
commit 35aa848168
143 changed files with 33558 additions and 0 deletions

View File

@@ -0,0 +1,19 @@
import { existsSync } from 'node:fs';
(function checkDevcontainer() {
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
if (inContainer) return;
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
if (process.env.CI === 'true' || process.env.CI === '1') return;
const yellow = '\x1b[33m';
const reset = '\x1b[0m';
process.stderr.write(
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
);
})();

View File

@@ -0,0 +1,92 @@
#!/bin/bash
# Read the harness registry and derive per-harness credentials from it.
#
# Source it — the whole point is exporting into the caller's environment, which a subshell
# would lose:
#
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
# harness_setup_credentials
#
# Two callers: `harbor-run`, which needs only this, and `setup-harnesses.sh`, which sources
# it and adds installs, config writing and launchers on top.
#
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
# post-creates run with -e). An unguarded failure below therefore aborts container
# creation, which is why every failure site is individually guarded rather than relying on
# this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
# the first one that can actually import it rather than assuming.
_raccoon_python() {
local p
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
[ -n "$p" ] || continue
command -v "$p" >/dev/null 2>&1 || continue
if "$p" -c "import tomllib" >/dev/null 2>&1; then
printf '%s' "$p"
return 0
fi
done
return 1
}
_harness_query() {
local py
py=$(_raccoon_python) || return 1
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url="${ANTHROPIC_BASE_URL:-}"
[ -n "$base_url" ] || return 1
base_url="${base_url%"${base_url##*[!/]}"}"
# ".../api/llm_proxy/raccoon" -> ".../api/llm_proxy". Requires a path to strip: a base
# URL that is a bare host with no path — a provider's own API root rather than the
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
case "${base_url#*://}" in
*/*) printf '%s' "${base_url%/*}" ;;
*) return 2 ;;
esac
}
harness_setup_credentials() {
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
# note at the top), and a bare failing assignment would exit the caller's post-create
# outright — silently, since the failure paths below are what do the explaining.
local root rc=0
root="$(_harness_proxy_root)" || rc=$?
if [ "$rc" -ne 0 ]; then
if [ "$rc" -eq 2 ]; then
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
echo "harness-setup: authenticated. Use the base URL you were given." >&2
else
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
fi
return 0
fi
local key="${ANTHROPIC_API_KEY:-}"
if [ -z "$key" ]; then
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
return 0
fi
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
if [ -z "${!key_env:-}" ]; then
export "$key_env=$key"
fi
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
export "$base_url_env=$root/$proxy_path"
fi
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}

View File

@@ -0,0 +1,391 @@
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
package.json has no zod/smol-toml).
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
copies.
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
sandbox agent, a plain unit test, or the devcontainer python alike.
"""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
@dataclass(frozen=True)
class Harness:
"""One harness, as declared in harness-registry.toml."""
id: str
label: str
agent_import_path: str
model_id_shape: str
writes_atif: bool
capture: bool
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
effort_kwarg: str = ""
effort_default: str | None = None
key_env: str | None = None
base_url_env: str | None = None
proxy_path: str | None = None
flaky_hangs: bool = False
enabled: bool = True
# Worker-container fields; see the registry header.
authoring: bool = False
cli: str | None = None
install: str | None = None
skills_dir: str | None = None
auth_path: str | None = None
auth_key_env: str | None = None
explore_launch: str | None = None
config_path: str | None = None
# Config the harness needs wherever it runs, trial sandbox included.
agent_config: str | None = None
# Config for both worker containers (explore and authoring).
container_config: str | None = None
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
# explore/plugins/. Writing them in authoring would register hooks against files that
# are not there, firing on every prompt.
explore_config: str | None = None
# Fields added for a later phase, kept verbatim so this loader doesn't have to
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there."""
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
def row_label(self, model: str) -> str:
"""Row identity for one trial: bare model for legacy harnesses (so
published manifests keep their labels), else ``<harness>:<model>``."""
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
def agent_config_overrides(self) -> dict[str, str]:
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
interpolates these into a shell command, which would strip the quotes anyway;
emitting them would only make the result depend on how many shell layers the
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
binary as ``model=o3``).
These settings ride the command line as ``-c dotted.key=value`` everywhere the
harness runs, never a config file. A trial sandbox rules the file out: the
harness's own runner appends root keys to it, and TOML has no way back to the
root scope once a table has opened, so a table we appended would swallow them.
Overrides compose in any order and beat the file, so the same rendering serves
the explore launcher too — one declaration, one mechanism.
"""
if not self.agent_config:
return {}
try:
parsed = tomllib.loads(self.agent_config)
except tomllib.TOMLDecodeError as exc:
raise HarnessRegistryError(
f"{self.id}: agent_config is not valid TOML ({exc})"
) from exc
flat: dict[str, str] = {}
def walk(node: dict, prefix: str) -> None:
for key, value in node.items():
path = f"{prefix}{key}"
if isinstance(value, dict):
walk(value, f"{path}.")
elif isinstance(value, bool):
flat[path] = "true" if value else "false"
elif isinstance(value, (int, float)):
flat[path] = str(value)
elif isinstance(value, str):
if value != value.strip() or any(c in value for c in " \"'\\"):
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has a value needing "
"shell quoting, which the -c override form cannot carry"
)
flat[path] = value
else:
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has type "
f"{type(value).__name__}, which has no -c override form"
)
walk(parsed, "")
return flat
def container_config_text(self, *, surface: str) -> str | None:
"""Config file body for a worker container. `surface` is "explore" or
"authoring"; explore additionally gets `explore_config`. Root keys come from
`container_config` first, so appending a table section stays valid TOML."""
parts = [self.container_config]
if surface == "explore":
parts.append(self.explore_config)
kept = [part.strip("\n") for part in parts if part and part.strip()]
return "\n\n".join(kept) + "\n" if kept else None
def agent_config_flags(self) -> str:
"""``agent_config`` as a ``-c key=value`` command-line string."""
return " ".join(
f"-c {key}={value}"
for key, value in sorted(self.agent_config_overrides().items())
)
def explore_launch_command(self) -> str | None:
"""``explore_launch`` with the registry's own values substituted in.
The worker's Explore session and the trial must run the same agent, so the
model, effort and reductions are declared once here and rendered into both.
A literal in the launch string would be a second declaration, and the two
would drift the first time one of them was updated alone.
Only these three placeholders are substituted; ``$@`` and
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
"""
if not self.explore_launch:
return None
return (
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
.replace("$RACCOON_MODEL", self.default_model or "")
.replace("$RACCOON_EFFORT", self.effort_default or "")
)
def known_import_paths(self) -> tuple[str, ...]:
paths = [self.agent_import_path, *self.import_path_aliases]
if self.agent_import_path_single_turn:
paths.append(self.agent_import_path_single_turn)
return tuple(paths)
_KNOWN_FIELDS = frozenset(
{
"id",
"label",
"agent_import_path",
"agent_import_path_single_turn",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
"model_id_shape",
"effort_kwarg",
"effort_default",
"key_env",
"base_url_env",
"proxy_path",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
"flaky_hangs",
"enabled",
"authoring",
"cli",
"install",
"skills_dir",
"auth_path",
"auth_key_env",
"explore_launch",
"config_path",
"agent_config",
"container_config",
"explore_config",
}
)
_REQUIRED_FIELDS = (
"id",
"label",
"agent_import_path",
"model_id_shape",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
)
class HarnessRegistryError(ValueError):
"""Malformed registry. Raised rather than tolerated: a broken registry is a
broken deployment, and silently defaulting would pick the wrong agent."""
def _references_agent_flags(launch: str) -> bool:
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
@dataclass(frozen=True)
class HarnessRegistry:
version: int
harnesses: tuple[Harness, ...]
def all(self) -> tuple[Harness, ...]:
return self.harnesses
def enabled(self) -> tuple[Harness, ...]:
return tuple(h for h in self.harnesses if h.enabled)
def authoring(self) -> tuple[Harness, ...]:
"""Harnesses a worker can author with — what the worker containers install.
Narrower than enabled(): a harness can be runnable in a trial without having
an authoring story (no CLI to converse with, or no capture)."""
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
def find(self, harness_id: str) -> Harness | None:
return next((h for h in self.harnesses if h.id == harness_id), None)
def require(self, harness_id: str) -> Harness:
harness = self.find(harness_id)
if harness is not None:
return harness
available = ", ".join(sorted(h.id for h in self.enabled()))
raise HarnessRegistryError(
f'Unknown harness "{harness_id}". Available: {available}'
)
def by_import_path(self, agent: str) -> Harness | None:
"""Resolve an agent identity — a ``name()`` or import path from
``result.json`` ``config.agent``, or a manifest row — to its harness."""
needle = (agent or "").strip()
if not needle:
return None
for harness in self.harnesses:
if needle == harness.id or needle in harness.known_import_paths():
return harness
return None
def _build(entry: dict, index: int) -> Harness:
for name in _REQUIRED_FIELDS:
if name not in entry:
raise HarnessRegistryError(
f"harness[{index}]: missing required field '{name}'"
)
shape = entry["model_id_shape"]
if shape not in MODEL_ID_SHAPES:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
f"{sorted(MODEL_ID_SHAPES)}"
)
# These three reach `eval` in setup-harnesses.sh, which is how they support the
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
# is ours, but "ours" is not an argument that survives a careless future edit.
for shell_field in ("config_path", "auth_path", "skills_dir"):
value = entry.get(shell_field)
if not isinstance(value, str):
continue
# A backtick or $( executes outright. A double quote closes the string these are
# interpolated into, and a semicolon then starts a new command inside it — same
# outcome, one step removed.
bad = [t for t in ("`", "$(", '"', ";") if t in value]
if bad:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): {shell_field} contains "
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
)
launch = entry.get("explore_launch")
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): declares agent_config but its "
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
"would then run with a different toolset than the trial it is authoring "
"for, which is the drift agent_config exists to prevent."
)
return Harness(
id=entry["id"],
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),
model_id_shape=shape,
effort_kwarg=entry.get("effort_kwarg", ""),
effort_default=entry.get("effort_default"),
key_env=entry.get("key_env"),
base_url_env=entry.get("base_url_env"),
proxy_path=entry.get("proxy_path"),
writes_atif=bool(entry["writes_atif"]),
capture=bool(entry["capture"]),
seed_native=bool(entry["seed_native"]),
seed_atif=bool(entry["seed_atif"]),
flaky_hangs=bool(entry.get("flaky_hangs", False)),
enabled=bool(entry.get("enabled", True)),
authoring=bool(entry.get("authoring", False)),
cli=entry.get("cli"),
install=entry.get("install"),
skills_dir=entry.get("skills_dir"),
auth_path=entry.get("auth_path"),
auth_key_env=entry.get("auth_key_env"),
explore_launch=entry.get("explore_launch"),
config_path=entry.get("config_path"),
agent_config=entry.get("agent_config"),
container_config=entry.get("container_config"),
explore_config=entry.get("explore_config"),
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
)
_cache: dict[Path, HarnessRegistry] = {}
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
file, a duplicate id, or an import path claimed by two harnesses (which would
make ``by_import_path`` depend on declaration order)."""
resolved = Path(path).resolve()
if resolved in _cache:
return _cache[resolved]
with open(resolved, "rb") as handle:
doc = tomllib.load(handle)
if "version" not in doc:
raise HarnessRegistryError("harness-registry: missing 'version'")
entries = doc.get("harness") or []
if not entries:
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
seen_ids: set[str] = set()
for harness in harnesses:
if harness.id in seen_ids:
raise HarnessRegistryError(
f"harness-registry: duplicate harness id: {harness.id}"
)
seen_ids.add(harness.id)
owners: dict[str, str] = {}
for harness in harnesses:
for import_path in harness.known_import_paths():
owner = owners.get(import_path)
if owner is not None and owner != harness.id:
raise HarnessRegistryError(
f'harness-registry: import path "{import_path}" claimed by both '
f'"{owner}" and "{harness.id}"'
)
owners[import_path] = harness.id
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
_cache[resolved] = registry
return registry

View File

@@ -0,0 +1,251 @@
/**
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
* that reference runs and detector reports depend on.
*
* A reference run is only meaningful for the task inputs it actually ran
* against: the prompt (instruction.md), the snapshot session
* (environment/session.jsonl), the workspace patch
* (environment/workspace.patch), and the gitref the workspace is built from
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
* revision of instruction.md + tests/grader-guidance.md. When any of those
* change after the artifact was produced, the artifact is stale — it describes
* an older revision of the task than the one being packaged.
*
* This module is the single source of truth for WHAT gets checksummed and how
* captures are compared. Capture sites (copy-reference-run.ts,
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
* the artifact; submit-task.ts re-captures at packaging time and diffs.
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
* mask an mtime, but can't change a sha256.
*
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
* occupies the same relative path there (mirroring the check-devcontainer
* pattern).
*
* Related but deliberately separate: `computeDeliveryHash` (repo-side
* delivery script — grep the internal repo for it; not shipped with the
* toolkit) hashes an overlapping input set for delivery idempotency. It is
* NOT built on this module because its hash format is load-bearing (a
* changed hash re-delivers every task); if you change WHAT counts as a task
* input here, check whether the delivery hash needs the same change.
*/
import { createHash } from 'node:crypto';
import { existsSync, readFileSync } from 'node:fs';
import { join } from 'node:path';
/** Bump when the record shape changes incompatibly. */
export const INPUT_CHECKSUMS_VERSION = 1;
/** Filename of the record inside a reference-run directory. */
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
/**
* Where in the artifact lifecycle a capture happened. The moment matters for
* how much a "fresh" verdict can be trusted:
*
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
* The strongest evidence: the record is what the agent ran
* against, whatever got edited afterwards.
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
* no run-time stamp. An input edited between harbor-run and the
* copy is recorded at its post-edit state, so a stale run can
* read fresh.
* - 'stamp' — record-detector-inputs.ts, right after a detector skill
* writes its report.
* - 'mirror' — fetch-detectors --write-dir, when canonical detector
* reports are re-materialized to disk from their remote
* store (repo-side flow only; never written in a worker
* checkout).
*
* Absent on records written before this field existed.
*/
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror';
/**
* The checksums of a task's inputs as they stood at capture time. Every hash
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
* that later becomes a hash — or vice versa — is a change like any other).
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
* than hashed so a mismatch message can show it.
*/
export interface TaskInputChecksums {
version: number;
/** ISO-8601 timestamp of the capture. */
capturedAt: string;
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
capturedBy?: TaskInputCaptureMethod;
/**
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
* by launch-time captures so stamping can be scoped to the right task's
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
* stamp is detectable after the fact.
*/
taskSlug?: string;
/**
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
* after the harbor-path scrub deliberately rewrote those docs
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
* task; only the doc bytes were normalized.
*/
restampedAt?: string;
inputs: {
prompt: string | null;
graderGuidance: string | null;
sessionJsonl: string | null;
workspacePatch: string | null;
gitref: string | null;
/**
* tests/grader-guidance-consolidated.md — the consolidated-standard
* guidance. Absent (undefined) on records captured before the field
* existed; comparisons skip a field the record predates, so old captures
* stay fresh until they are re-stamped. Last in field order because the
* task-checksum digest serializes fields in this order and appends new
* fields at the end (see scripts/lib/grader-run-checksums.ts).
*/
graderGuidanceConsolidated?: string | null;
};
}
export type TaskInputName = keyof TaskInputChecksums['inputs'];
/** Human-readable component names, used verbatim in staleness warnings. */
export const INPUT_LABELS: Record<TaskInputName, string> = {
prompt: 'prompt (instruction.md)',
graderGuidance: 'grader guidance (tests/grader-guidance.md)',
graderGuidanceConsolidated:
'consolidated grader guidance (tests/grader-guidance-consolidated.md)',
sessionJsonl: 'session snapshot (environment/session.jsonl)',
workspacePatch: 'workspace patch (environment/workspace.patch)',
gitref: 'gitref (task.toml commit)',
};
/**
* The inputs that shape what the AGENT saw and did. Changing any of them means
* a captured reference run no longer reflects the task being packaged, and
* only re-running the agent can fix that. The grader-guidance files (legacy
* and consolidated) are deliberately NOT in this set: editing the rubric
* stales the run's GRADE, not the run itself, and `scripts/harbor-regrade`
* re-derives grades without re-running the agent.
*/
export const REFERENCE_RUN_INPUTS: readonly TaskInputName[] = [
'prompt',
'sessionJsonl',
'workspacePatch',
'gitref',
];
/**
* The inputs a detector report assesses — instruction.md plus whichever
* grader-guidance files the task carries (the legacy pair is what the old
* mtime comparison in submit-task.ts watched; the consolidated file joined
* when grading moved to the consolidated standard). Compared by content.
*/
export const DETECTOR_REPORT_INPUTS: readonly TaskInputName[] = [
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
];
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
}
/**
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
* missing, unreadable, or has no commit line — a read failure downgrades to
* "absent" rather than crashing a capture or a validation sweep.
*
* Deliberately a regex, not a TOML parser: this module ships in the worker
* toolkit, where a new runtime dep would break packaging for every worker
* whose container predates the dep (npm install runs only on container
* create, and containers survive toolkit upgrades). build-workspace.sh reads
* the same key with the same grep-a-`commit`-line approach. The one `commit`
* key in a task.toml is `[metadata].commit`, so anchoring to the first
* `commit = "…"` line is exact in practice.
*/
function readGitref(taskDir: string): string | null {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return null;
try {
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
readFileSync(tomlPath, 'utf-8')
);
const commit = match?.[1] ?? match?.[2];
return commit && commit.length > 0 ? commit : null;
} catch {
return null;
}
}
/** Checksum the task inputs as they stand right now under `taskDir`. */
export function captureTaskInputs(
taskDir: string,
capturedBy?: TaskInputCaptureMethod
): TaskInputChecksums {
return {
version: INPUT_CHECKSUMS_VERSION,
capturedAt: new Date().toISOString(),
...(capturedBy ? { capturedBy } : {}),
inputs: {
prompt: sha256File(join(taskDir, 'instruction.md')),
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
gitref: readGitref(taskDir),
graderGuidanceConsolidated: sha256File(
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
),
},
};
}
/**
* Read a previously captured record. Returns null when the file is missing or
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
* future incompatible version) — callers treat null as "staleness unknowable",
* never as an error. No zod here: this module ships in the worker toolkit,
* whose dependency set stays minimal, so the guard is manual.
*/
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
if (!existsSync(filePath)) return null;
let parsed: unknown;
try {
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
} catch {
return null;
}
if (typeof parsed !== 'object' || parsed === null) return null;
const record = parsed as TaskInputChecksums;
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
const value = record.inputs[name];
// undefined = the record predates this input field; still a valid capture.
if (value !== undefined && value !== null && typeof value !== 'string') return null;
}
return record;
}
/**
* Which of `names` changed between a recorded capture and the current state?
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
* whose value differs — including absent→present and present→absent flips. A
* field the recorded capture predates (the key is not in the record at all)
* is skipped: freshness on that axis is unknowable, and flagging every old
* record the moment a new axis ships would drown the real signal.
*/
export function diffTaskInputs(
recorded: TaskInputChecksums,
current: TaskInputChecksums,
names: readonly TaskInputName[]
): string[] {
return names
.filter((name) => name in recorded.inputs)
.filter((name) => recorded.inputs[name] !== current.inputs[name])
.map((name) => INPUT_LABELS[name]);
}

View File

@@ -0,0 +1,399 @@
/**
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
*
* `environment/Dockerfile`, `tests/test.sh`, `tests/grader-system-prompt.md`
* and `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
* are the same in every task: they decide how the trial runs and how the grade
* is produced. An edit makes a task's reference runs incomparable to every
* other task's, and the scores still look normal, so nothing downstream
* notices.
*
* A task is compared against itself as created. {@link writeManagedStamp} records
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
* creation, so a later mismatch is an edit made since. Tasks created before
* stamping have no record and fall back to matching the copies this toolkit
* ships — see {@link IntegrityStatus}.
*
* The toolkit appends to a task's Dockerfile itself (session staging, the
* reference-data corpus). Those blocks are wrapped in
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
* comparing, so they never read as edits.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { basename, join } from 'path';
/**
* Every status is advisory. Nothing here stops a trial or a submission: an author
* who changed one of these files did it because they didn't know we'd rather they
* didn't, and refusing to package their work punishes a misunderstanding. The job
* is to say so clearly, and to record it so a reviewer sees it too.
*
* `ok` — identical to a copy this toolkit ships, or unchanged since the
* task was created.
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
* copy since. Nobody's mistake; it does mean this task's runs
* aren't directly comparable to one built today.
* `modified` — matches neither its baseline nor anything shipped: an edit.
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
* and an older release are indistinguishable.
* `missing` — the task doesn't have the file.
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
* been selected yet.
*/
export type IntegrityStatus =
| 'ok'
| 'outdated'
| 'modified'
| 'unverifiable'
| 'missing'
| 'placeholder';
export interface FileVerdict {
/** Task-relative path, e.g. `environment/Dockerfile`. */
taskPath: string;
status: IntegrityStatus;
/** Command that restores the managed version, on `modified` / `unverifiable`. */
restore?: string;
}
export interface IntegrityReport {
/** False when this isn't a worker toolkit — callers should skip silently. */
checked: boolean;
files: FileVerdict[];
/** Looks like an edit: matches neither a baseline nor anything shipped. */
modified: FileVerdict[];
/** Can't be told apart from an older release. */
unverifiable: FileVerdict[];
/** Unchanged, but a newer copy has shipped since. */
outdated: FileVerdict[];
}
interface ManagedFile {
taskPath: string;
/** Matches the candidate pristine filenames under `task-shared/`. */
baselinePattern: RegExp;
}
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
const MANAGED_FILES: ManagedFile[] = [
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
{ taskPath: 'tests/grader-system-prompt.md', baselinePattern: /^grader-system-prompt\.md$/ },
{
taskPath: 'tests/grader-system-prompt-consolidated.md',
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
},
];
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
/**
* Line shapes from toolkit releases that predate the sentinels. Deliberately
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
* lines" rule an edit could hide behind.
*/
const LEGACY_MANAGED_LINES: RegExp[] = [
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
];
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
/** Per-task stamp of the managed files as created. Lives in the task directory. */
export const STAMP_FILENAME = '.toolkit-managed.json';
interface ManagedStamp {
version: number;
stampedAt: string;
/** taskPath → sha256 of the stripped content. */
files: Record<string, string>;
}
/**
* Remove toolkit-appended content so only author-authored differences remain.
* Trailing blank lines go too — an editor adding or trimming a final newline is
* not something to fail a trial over.
*/
export function stripManagedBlocks(content: string): string {
const out: string[] = [];
let inBlock = false;
// Normalize CRLF before anything else: a Windows editor or a checkout with
// core.autocrlf rewrites every line ending, and that must not read as an edit.
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
if (!inBlock && SENTINEL_OPEN.test(line)) {
inBlock = true;
continue;
}
if (inBlock) {
if (SENTINEL_CLOSE.test(line)) inBlock = false;
continue;
}
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
out.push(line);
}
return out.join('\n').replace(/\s+$/, '');
}
export function sha256(content: string): string {
return createHash('sha256').update(content).digest('hex');
}
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
if (!existsSync(sharedDir)) return [];
return readdirSync(sharedDir)
.filter((f) => pattern.test(f))
.sort();
}
/**
* Record the managed files, so later edits are detectable. Call at task creation
* and after a managed file is first put in place.
*
* A file earns a baseline only by matching a copy this toolkit ships, and an
* entry already recorded is never rewritten. Together those mean a stamp can
* only ever describe a pristine file: re-running this can't turn an author's
* edit into the new baseline, and a file dropped in later (the polyglot
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
* baseline once it's in place.
*
* Returns true if anything was recorded.
*/
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
const sharedDir = join(toolkitRoot, 'task-shared');
const existing = readStamp(taskDir);
const files: Record<string, string> = { ...(existing?.files ?? {}) };
let added = false;
for (const managed of MANAGED_FILES) {
if (files[managed.taskPath]) continue;
const p = join(taskDir, managed.taskPath);
if (!existsSync(p)) continue;
const raw = readFileSync(p, 'utf-8');
// Not a baseline: the author still has to drop in their member's base image.
if (raw.includes(PLACEHOLDER_MARKER)) continue;
const stripped = stripManagedBlocks(raw);
if (!matchesShipped(sharedDir, managed, stripped)) continue;
files[managed.taskPath] = sha256(stripped);
added = true;
}
if (!added) return false;
const stamp: ManagedStamp = {
version: 1,
stampedAt: new Date().toISOString(),
files,
};
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
return true;
}
function readStamp(taskDir: string): ManagedStamp | null {
const stampPath = join(taskDir, STAMP_FILENAME);
if (!existsSync(stampPath)) return null;
try {
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
return null;
}
}
/**
* How to restore a managed file, or undefined when this toolkit ships no copy to
* restore from. Only the Dockerfile can have several candidates (one per member).
*/
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
const dest = `harbor-tasks/${slug}/${taskPath}`;
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
if (candidates.length > 1) {
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
}
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
// author to copy a Dockerfile over their grader prompt.
return undefined;
}
/** Render a restore line, saying so plainly when there is nothing to restore from. */
function restoreLine(f: FileVerdict): string {
return f.restore
? ` ${f.restore}`
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
}
/** Does this content match a pristine copy the toolkit ships? */
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
return baselineCandidates(sharedDir, managed.baselinePattern).some(
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
);
}
/**
* Compare a task's managed files against its creation-time stamp.
*
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
*/
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
const sharedDir = join(toolkitRoot, 'task-shared');
// Without task-shared/ there is nothing to compare against; report "not
// checked" so callers no-op rather than reporting three phantom failures.
if (!existsSync(sharedDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
const slug = basename(taskDir);
const stamp = readStamp(taskDir);
const files: FileVerdict[] = [];
for (const managed of MANAGED_FILES) {
const taskFile = join(taskDir, managed.taskPath);
if (!existsSync(taskFile)) {
files.push({ taskPath: managed.taskPath, status: 'missing' });
continue;
}
const raw = readFileSync(taskFile, 'utf-8');
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
const restore = restoreCommand(managed.taskPath, candidates, slug);
const stripped = stripManagedBlocks(raw);
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
// cannot be an author edit, whatever the stamp says — and asking the stamp first
// is what used to make restoring the current copy (which is exactly what we tell
// authors to do) look like an edit, with no way out.
if (matchesShipped(sharedDir, managed, stripped)) {
files.push({ taskPath: managed.taskPath, status: 'ok' });
continue;
}
const expected = stamp?.files[managed.taskPath];
if (expected) {
// Matches its baseline but nothing shipped: untouched by the author, and the
// toolkit has moved on since. Worth saying, nobody's fault.
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
files.push({ taskPath: managed.taskPath, status, restore });
continue;
}
// Checked after the stamp so that adding this marker to a file that HAS a
// baseline can't exempt it from the comparison.
if (raw.includes(PLACEHOLDER_MARKER)) {
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
continue;
}
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
}
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
unverifiable: files.filter((f) => f.status === 'unverifiable'),
outdated: files.filter((f) => f.status === 'outdated'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal. An author who changed one of these files
* almost always did it to get unstuck, not knowing we'd rather they told us — so
* this explains what it means for their task and what restoring would do, and then
* lets them get on with it.
*/
export function formatIntegrityReport(report: IntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These files look edited since this task was created, and the toolkit manages',
'them — they set up how the trial runs and how the grade is produced, so they',
"have to be identical across every task. Yours aren't, which makes this task's",
"runs hard to compare with everyone else's:",
'',
...report.modified.map((f) => ` ${f.taskPath}`),
'',
'Restoring the shipped version puts that right:',
...report.modified.map(restoreLine),
'',
'If you changed one to work around a problem — a missing package, a grader that',
"wouldn't run — please tell us about the problem instead. It almost certainly",
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.outdated.length > 0) {
sections.push(
[
'These files are unchanged, but the toolkit has shipped newer copies since this',
'task was created:',
'',
...report.outdated.map((f) => ` ${f.taskPath}`),
'',
"You haven't done anything wrong. It does mean this task was run and graded with",
"older versions than a task built today, so its scores aren't directly",
'comparable. To line them up, restore the current copies and re-run your trials:',
...report.outdated.map(restoreLine),
].join('\n')
);
}
if (report.unverifiable.length > 0) {
sections.push(
[
"These files don't match the copies this toolkit ships, and this task has no",
'record of what they looked like when it was created:',
'',
...report.unverifiable.map((f) => ` ${f.taskPath}`),
'',
'Two things look like this and we cannot tell them apart: a task created on an',
'earlier toolkit release (nothing to fix, though its scores are not directly',
'comparable to a task built today), or a file that was edited. Either way,',
'restoring the current copy and re-running your trials is what makes this task',
"comparable to everyone else's:",
...report.unverifiable.map(restoreLine),
].join('\n')
);
}
return sections.join('\n\n');
}
/**
* Wrap a report in a banner loud enough to survive a scrollback.
*
* Nothing blocks any more, so this notice is the entire mechanism — and an
* unframed paragraph among build output is one a reasonable person scrolls past.
* Yellow only when stderr is a terminal, so piped logs stay clean.
*/
export function bannerize(message: string, report: IntegrityReport): string {
const RULE = '#'.repeat(78);
const headline =
report.modified.length > 0
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!';
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
return `${color[0]}${body}${color[1]}`;
}

View File

@@ -0,0 +1,241 @@
/**
* Tests for tree-permissions.ts.
*
* The load-bearing case is the one from the field report: a directory that came
* across without its search bit makes `tar` fail with `Cannot stat` on the files
* *inside* it, so the repair has to fix directory modes, not just ownership.
* These tests run unprivileged, so they exercise the mode axis for real and the
* ownership axis only as far as an unprivileged process can (target resolution +
* graceful EPERM), which is the same shape CI runs in.
*/
import assert from 'node:assert/strict';
import { chmodSync, mkdirSync, rmSync, statSync, symlinkSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { test } from 'node:test';
import {
didRepair,
manualRepairHint,
normalizeTreePermissions,
resolveWorkspaceOwner,
} from './tree-permissions';
function scratch(name: string): string {
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
rmSync(dir, { recursive: true, force: true });
mkdirSync(dir, { recursive: true });
return dir;
}
test('restores the search bit on a directory that lost it', () => {
const root = scratch('searchbit');
const models = join(root, 'agent-output', 'app', 'models');
mkdirSync(models, { recursive: true });
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
// r-- : readdir works, so tar can NAME the file, but stat is refused.
chmodSync(models, 0o400);
const report = normalizeTreePermissions(root);
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
assert.ok(report.modeFixed.some((p) => p === models));
assert.ok(didRepair(report));
rmSync(root, { recursive: true, force: true });
});
test('recurses into a directory it had to widen first', () => {
const root = scratch('recurse');
const inner = join(root, 'locked', 'deeper');
mkdirSync(inner, { recursive: true });
const leaf = join(inner, 'leaf.rb');
writeFileSync(leaf, 'x\n');
chmodSync(leaf, 0o000);
chmodSync(inner, 0o400);
chmodSync(join(root, 'locked'), 0o400);
const report = normalizeTreePermissions(root);
// Only reachable if the walk widened each parent before descending.
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
assert.ok(report.modeFixed.includes(leaf));
rmSync(root, { recursive: true, force: true });
});
test('leaves already-correct trees untouched', () => {
const root = scratch('noop');
mkdirSync(join(root, 'sub'), { recursive: true });
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
const report = normalizeTreePermissions(root);
assert.deepEqual(report.modeFixed, [], 'no mode changes');
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
rmSync(root, { recursive: true, force: true });
});
test('does not widen group/other beyond what was already there', () => {
const root = scratch('narrow');
const f = join(root, 'secret.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
normalizeTreePermissions(root);
const mode = statSync(f).mode & 0o777;
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
rmSync(root, { recursive: true, force: true });
});
test('ignores symlinks rather than following them out of the tree', () => {
const root = scratch('symlink');
const outside = scratch('symlink-outside');
const victim = join(outside, 'victim.txt');
writeFileSync(victim, 'x\n');
chmodSync(victim, 0o000);
symlinkSync(outside, join(root, 'link'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
rmSync(outside, { recursive: true, force: true });
});
test('never throws on a missing root, and reports it', () => {
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
assert.equal(report.failures.length, 1);
assert.equal(report.failures[0].reason, 'ENOENT');
});
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
const root = scratch('owner');
const owner = resolveWorkspaceOwner(root);
assert.ok(owner, 'resolved');
const st = statSync(root);
assert.equal(owner.uid, st.uid);
assert.equal(owner.gid, st.gid);
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
rmSync(root, { recursive: true, force: true });
});
test('never chowns TO root, even when the owner ref is root-owned', () => {
// The regression this guards: workspace root owned by root (unzipped with
// sudo) while the task files are correctly owned by the human. Chowning to the
// ref's owner would inflict the very lockout this module prevents. `/` is
// root-owned on every platform we run on, so it's a stable stand-in.
const root = scratch('root-ref');
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
const beforeUid = statSync(f).uid;
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(report.target?.uid, 0, 'resolved a root target');
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
assert.deepEqual(report.failures, [], 'and did not fail trying');
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
rmSync(root, { recursive: true, force: true });
});
test('still normalizes modes when the chown target is root', () => {
const root = scratch('root-ref-modes');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
writeFileSync(join(sub, 'f.txt'), 'x\n');
chmodSync(sub, 0o400);
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
assert.ok(report.modeFixed.includes(sub));
rmSync(root, { recursive: true, force: true });
});
test('walks a tree as deep as the filesystem allows', () => {
const root = scratch('deep');
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
// call-stack limit. So this isn't a stack test; it just pins that a deep,
// narrow tree walks cleanly end to end.
let path = root;
for (let i = 0; i < 250; i++) {
path = join(path, `d${i}`);
}
mkdirSync(path, { recursive: true });
writeFileSync(join(path, 'leaf.txt'), 'x\n');
chmodSync(join(path, 'leaf.txt'), 0o000);
const report = normalizeTreePermissions(root);
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
rmSync(root, { recursive: true, force: true });
});
test('a failure in one subtree does not abandon the rest', () => {
const root = scratch('partial');
const good = join(root, 'good');
mkdirSync(good, { recursive: true });
const goodFile = join(good, 'f.txt');
writeFileSync(goodFile, 'x\n');
chmodSync(goodFile, 0o000);
// A dangling symlink and a vanished path both produce per-entry trouble.
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
assert.ok(report.modeFixed.includes(goodFile));
rmSync(root, { recursive: true, force: true });
});
test('reports rather than throws when the root is a file, not a directory', () => {
const root = scratch('file-root');
const f = join(root, 'lonely.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
const report = normalizeTreePermissions(f);
assert.equal(statSync(f).mode & 0o600, 0o600);
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
});
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
const root = scratch('killswitch');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'f.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
chmodSync(sub, 0o400);
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
try {
const report = normalizeTreePermissions(root);
assert.equal(report.skipped, true);
assert.deepEqual(report.modeFixed, []);
assert.deepEqual(report.ownerFixed, []);
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
} finally {
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
}
chmodSync(sub, 0o700);
rmSync(root, { recursive: true, force: true });
});
test('manual hint repairs both axes, ownership first', () => {
const hint = manualRepairHint('harbor-tasks/my-slug');
assert.match(hint, /chown -R/);
assert.match(hint, /chmod -R u\+rwX/);
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
});

View File

@@ -0,0 +1,119 @@
/**
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
*
* Files captured from a task run can arrive owned by another user, or with a
* directory missing the permission needed to walk into it. Packaging then fails
* with `Cannot stat: Permission denied`. This repairs both.
*
* Grants owner rwX only, never group or other. Never throws, and never hands
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
*/
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
import { join } from 'path';
export interface NormalizeReport {
/** Paths whose owner was changed. */
ownerFixed: string[];
/** Paths whose mode gained owner rwX. */
modeFixed: string[];
/** Paths we wanted to change but could not, with the errno. */
failures: { path: string; reason: string }[];
/** Resolved target owner, or null if it couldn't be determined. */
target: { uid: number; gid: number } | null;
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
skipped?: boolean;
}
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
try {
const st = statSync(ownerRef);
return { uid: st.uid, gid: st.gid };
} catch {
return null;
}
}
/** Owner-rwX mode, preserving every other bit. Dirs also need the search bit. */
function withOwnerAccess(mode: number, isDir: boolean): number {
return mode | (isDir ? 0o700 : 0o600);
}
/**
* Give every entry under `root` to the workspace owner and make sure that owner
* can read and traverse it. Symlinks are skipped. Repairs what it can and
* reports what it couldn't; it never throws and never blocks its caller.
*/
export function normalizeTreePermissions(
root: string,
options: { ownerRef?: string } = {}
): NormalizeReport {
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
}
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
// Never hand files to root — that would lock the owner out rather than help.
const chownTarget = target && target.uid !== 0 ? target : null;
try {
const stack: string[] = [root];
while (stack.length > 0) {
const path = stack.pop() as string;
let st;
try {
st = lstatSync(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
continue;
}
if (st.isSymbolicLink()) continue;
const isDir = st.isDirectory();
// Mode first: a directory we can't search is one we can't descend into.
const wanted = withOwnerAccess(st.mode, isDir);
if (wanted !== st.mode) {
try {
chmodSync(path, wanted);
report.modeFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
}
}
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
try {
chownSync(path, chownTarget.uid, chownTarget.gid);
report.ownerFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
}
}
if (!isDir) continue;
try {
for (const entry of readdirSync(path)) stack.push(join(path, entry));
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
}
}
} catch (err) {
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
}
return report;
}
/** True when something was actually repaired. */
export function didRepair(report: NormalizeReport): boolean {
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
}
/** The command to run on your host if we couldn't fix it ourselves. */
export function manualRepairHint(path: string): string {
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
}