ren worker folder adding orig, mv new one into root

This commit is contained in:
2026-09-25 10:34:29 -04:00
parent 10f0668e32
commit 5b010039d7
1308 changed files with 44597 additions and 1511 deletions

View File

@@ -0,0 +1,34 @@
# shellcheck shell=bash
# call-origin.sh — build the X-Surge-Client-Metadata header value.
#
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
# proxy routes: those are platform-internal, this is not, so this file is the
# half that ships in the worker toolkit — worker runs go through the same proxy
# and are attributed the same way.
#
# . scripts/lib/call-origin.sh
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
#
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
# invalid_client_metadata), so an origin that is not a plain slug yields an
# empty string and the caller sends no header at all: losing attribution beats
# failing the call.
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
# An unlabelled call is still a real call, so it gets a bucket rather than no
# header: a missing origin in the audit log then means an unplumbed surface.
DEFAULT_CALL_ORIGIN="local"
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
# Only a slug matching this pattern is ever interpolated, so nothing needs
# JSON-escaping and this stays dependency-free (it is sourced in worker
# containers, which have no python).
call_origin_metadata() {
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
case "$origin" in
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
esac
[ "${#origin}" -le 64 ] || return 0
printf '{"origin":"%s"}' "$origin"
}

View File

@@ -190,16 +190,39 @@ if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1
raise SystemExit(1)
text = os.path.expandvars(text)
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
# root. Every other table, [hooks] on the explore surface included, is left alone.
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
wanted = []
section = None
for line in text.splitlines():
if line.lstrip().startswith("["):
break
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
stripped = line.strip()
if stripped.startswith("["):
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
continue
if section is False:
continue
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
if m:
wanted.append((m.group(1), line.rstrip()))
wanted.append((section, m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
def section_path(header):
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
return tuple(header.strip("[]").split("."))
def lookup(doc, header, key):
"""The value a parsed config holds for a wanted key, or KeyError."""
node = doc
if header:
for part in section_path(header):
node = node[part]
return node[key]
mode = None
if os.path.exists(target):
try:
@@ -208,23 +231,52 @@ if os.path.exists(target):
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
# Everything from the first table header on belongs to a table. A key appended after
# one is reparented into it, so both the search and the insert stay above the line.
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
changed = False
for key, line in wanted:
# The quoted spelling is the same key: replacing it beats adding a duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
def span(header):
"""The line range a section owns, or None when the file has no such section.
Root is everything above the first table header: a key appended below one
would be reparented into it, so searches and inserts stay inside the span.
"""
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
if header is None:
return 0, (heads[0] if heads else len(lines))
at = next((i for i in heads if lines[i].strip() == header), None)
if at is None:
if root_end < len(lines) and lines[root_end].strip():
lines.insert(root_end, "")
lines.insert(root_end, line)
root_end += 1
changed = True
elif lines[at] != line:
lines[at] = line
return None
after = next((i for i in heads if i > at), len(lines))
return at + 1, after
# Grouped, root first, so a section this file lacks can be written whole.
grouped = {}
for header, key, line in wanted:
grouped.setdefault(header, []).append((key, line))
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
changed = False
for header in ordered:
if span(header) is None:
# A config written before this section existed. Write the whole table
# rather than leave a root key naming a provider that is not there.
if lines and lines[-1].strip():
lines.append("")
lines.append(header)
lines.extend(line for _, line in grouped[header])
changed = True
continue
for key, line in grouped[header]:
# Re-read the span: an insert for an earlier key moved it.
start, end = span(header)
# The quoted spelling is the same key: replace rather than duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
if at is None:
if end < len(lines) and lines[end].strip():
lines.insert(end, "")
lines.insert(end, line)
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
@@ -238,9 +290,22 @@ try:
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have reached the root.
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
raise SystemExit(1)
# parses while leaving the key unset. Require every key to have landed on the value the
# blob asks for, in its own section — skipping sections this file does not carry.
blob_doc = tomllib.loads(text)
for header, key, _ in wanted:
try:
expected = lookup(blob_doc, header, key)
except (KeyError, TypeError):
raise SystemExit(1)
try:
got = lookup(doc, header, key)
except (KeyError, TypeError):
if header is None:
raise SystemExit(1)
continue
if got != expected:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())

View File

@@ -193,21 +193,48 @@ export const REFERENCE_RUN_INPUTS = Object.freeze([
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
* before the rename, plus the legacy-era plain-named file when a task
* authored on an earlier generation carries one), plus the atomic-rubric
* files the rubric detectors assess (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md). An absent file hashes to
* authored on an earlier generation carries one). An absent file hashes to
* null on both sides and never diffs. Compared by content.
*
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
* seventeen detectors never open them, so writing an atomic rubric after
* running the detectors would stale every one of those reports over files
* they never read. The two that do read them use
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
*/
export const DETECTOR_REPORT_INPUTS = Object.freeze([
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
'holisticRubric',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
* against the holistic rubric.
*/
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
...DETECTOR_REPORT_INPUTS,
'atomicRubric',
'rubricsYaml',
'graderContext',
] as const satisfies readonly TaskInputName[]);
/** Detectors that read the atomic-rubric package, and so are staled by it. */
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
'detector-rubric-coverage',
'detector-rubric-form',
]);
/** The input set a named detector's report is judged against. */
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
? RUBRIC_DETECTOR_REPORT_INPUTS
: DETECTOR_REPORT_INPUTS;
}
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;

View File

@@ -0,0 +1,96 @@
# shellcheck shell=bash
#
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
# translating through a commit map when history has been rewritten under it.
#
# A task pins a commit in task.toml. If that repo's history is later rewritten
# (to strip something that should never have shipped, say), every rewritten
# commit gets a new SHA and the pin stops resolving — including on machines we
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
# at repo/, so the directory name is not the member name and cannot be the key.
#
# The map is only consulted when the pin does not resolve, so it carries only
# rewritten commits — an unchanged commit resolves on its own and its identity
# row could never be read.
#
# Usage (source, then call):
# . "$(dirname "$0")/lib/resolve-pin.sh"
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
#
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
# pin cannot be resolved, having explained why.
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
_resolve_pin_hop() {
local from="$1" map="$2" hits
hits=$(awk -v p="$from" '
/^#/ || NF < 2 { next }
index($1, p) == 1 { print $2 }
' "$map" | sort -u)
[ -z "$hits" ] && return 1
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
return 2
fi
printf '%s\n' "$hits"
}
resolve_pin() {
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
# Present in the repo: nothing to translate.
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
printf '%s\n' "$sha"
return 0
fi
map=""
if [ -n "$map_dir" ]; then
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
map="$map_dir/$key.map"
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
map="$map_dir/$(basename "$repo_dir").map"
fi
fi
if [ -z "$map" ]; then
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
echo " The repo may be a shallow or partial copy — try a full clone." >&2
return 1
fi
# Follow the chain: a commit rewritten more than once maps forward a hop per
# rewrite, so keep going until the SHA exists or the trail ends.
cur="$commit"
hops=0
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
[ "$rc" -eq 2 ] && return 1
if [ "$rc" -ne 0 ]; then
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
return 1
fi
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
echo " Re-pin this task to a commit that still exists." >&2
return 1
fi
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
echo " Pin $commit was rewritten; using $sha" >&2
printf '%s\n' "$sha"
return 0
fi
cur="$next"
hops=$((hops + 1))
done
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
echo " $(basename "$map") may contain a cycle." >&2
return 1
}