ren worker folder adding orig, mv new one into root

This commit is contained in:
2026-09-25 10:34:29 -04:00
parent 10f0668e32
commit 5b010039d7
1308 changed files with 44597 additions and 1511 deletions

View File

@@ -72,7 +72,7 @@ RUN curl -fsSL https://pgp.mongodb.com/server-8.0.asc \
# (3.11+), and post-create runs under `set -e` — an image with only 3.10 fails container
# creation. setup-harnesses.sh tries python3, then python3.13/3.12/3.11, so exposing the
# newer one under its versioned name is enough and leaves the members' default untouched.
RUN curl -fsSL https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh \
RUN curl -fsSL https://astral.sh/uv/0.12.10/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh \
&& uv python install 3.10 \
&& ln -sf "$(uv python find 3.10)" /usr/local/bin/python3 \
&& uv python install 3.11 \

View File

@@ -0,0 +1,24 @@
#!/bin/bash
# Trial parity: limit DNS to the model endpoint and the toolkit's telemetry, so a session
# captured here cannot depend on network the trial agent will not have. Opt-in, best-effort.
#
# Its own lifecycle step, NOT part of post-start.sh: post-create.sh calls post-start to get
# postgres up before it installs the repo's dependencies, so a jail applied there breaks
# `bundle install` on every fresh container. postStartCommand runs after postCreateCommand.
set -u
[ -f /workspace/.devcontainer/dns-jail.sh ] || exit 0
# Read the one line rather than sourcing: with the jail off this must not execute the
# worker's .env as a side effect.
if [ "${RACCOON_DNS_JAIL:-0}" != "1" ] &&
! grep -qE '^[[:space:]]*(export[[:space:]]+)?RACCOON_DNS_JAIL[[:space:]]*=[[:space:]]*"?1"?[[:space:]]*(#.*)?$' \
/workspace/.env 2>/dev/null; then
exit 0
fi
(
set -a
# shellcheck disable=SC1091
. /workspace/.env 2>/dev/null || true
set +a
bash /workspace/.devcontainer/dns-jail.sh
) || true

View File

@@ -4,7 +4,7 @@
"build": {
"dockerfile": "Dockerfile",
"args": {
"TOOLKIT_BUILD_ID": "1788802488308-63ncdn"
"TOOLKIT_BUILD_ID": "1789430760274-eddbpv"
}
},
"appPort": [
@@ -21,6 +21,6 @@
"source=${localWorkspaceFolder}/repos${localEnv:EXPLORE_INSTANCE:},target=/workspace/repos,type=bind"
],
"postCreateCommand": "bash /workspace/.devcontainer/post-create.sh",
"postStartCommand": "bash /workspace/.devcontainer/post-start.sh",
"postStartCommand": "bash /workspace/.devcontainer/post-start.sh; bash /workspace/.devcontainer/apply-dns-jail.sh",
"containerUser": "root"
}

View File

@@ -366,13 +366,25 @@ mkdir -p "$HOME/.claude"
# SKIP_FAST_MODE_NETWORK_ERRORS: the LLM proxy doesn't forward claude's fast-mode
# availability probe, and claude reads the failed probe as "no network" and refuses
# /fast. The override makes /fast toggleable; fast serving stays OFF until toggled.
node -e '
# Names this container as the surface a proxy call came from: claude reads it from the
# settings env below, codex from the shell (its config maps the header to this var name).
# Only on the proxy — a provider endpoint must not carry this attribution.
CALL_METADATA=""
case "$(set -a; . /workspace/.env 2>/dev/null || true; set +a; printf '%s' "${ANTHROPIC_BASE_URL:-}")" in
*/llm_proxy/*) CALL_METADATA='{"origin":"explore"}' ;;
esac
CALL_METADATA="$CALL_METADATA" node -e '
const fs = require("fs");
const home = process.env.HOME;
const env = {
CLAUDE_CODE_DISABLE_AUTO_MEMORY: "1",
CLAUDE_CODE_SKIP_FAST_MODE_NETWORK_ERRORS: "1",
};
if (process.env.CALL_METADATA) {
env.ANTHROPIC_CUSTOM_HEADERS =
`X-Surge-Client-Metadata: ${process.env.CALL_METADATA}`;
}
fs.writeFileSync(
`${home}/.claude/settings.json`,
JSON.stringify({ env }, null, 2) + "\n"
@@ -388,6 +400,13 @@ if [ -d /workspace/data/zeta-corpus ]; then
fi
# Shell setup
# Ahead of the block below so every shell exports it, interactive or not.
if [ -n "$CALL_METADATA" ]; then
# A file rather than a .bashrc line alone: the launcher and refresh-harness-auth
# read it too, so a non-login shell does not silently lose the attribution.
printf 'export LLM_CALL_METADATA=%s\n' "'$CALL_METADATA'" > ~/.raccoon-call-origin
printf '. "$HOME/.raccoon-call-origin"\n' >> ~/.bashrc
fi
cat >> ~/.bashrc <<'BASHRC'
export PATH="$HOME/.local/bin:$PATH"
set -a && source /workspace/.env && set +a
@@ -407,8 +426,8 @@ bash /workspace/welcome.sh explore 2>/dev/null
_AK="fde503c3bdb6e5cc9c48b1f8e4c2abeb"
_DK="e966e45af5ad1a18005f9fdb831186ea"
_WID="w-mtriw5pe-u8me"
_VER="2f696c53b4"
_WID="w-mu1wvj9k-gtnk"
_VER="037cfcf94b"
_CT="explore"
_RP=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null)
_SID="$(date +%s)-$$"

View File

@@ -59,29 +59,6 @@ wait_for_pg() {
# Polyglot toolkit: REPO_NAME is empty (no single repo). Bring up Postgres + Redis
# (members need them; per-member DB setup is deferred to run-app/setup_repo), then done.
# Trial parity: limit DNS to the model endpoint and the toolkit's telemetry, so a session
# captured here cannot depend on network the trial agent will not have. Opt-in
# (RACCOON_DNS_JAIL=1) and best-effort. postCreate patches .bashrc, but postStart gets no
# login shell, so .env is read directly. Called on BOTH paths: the polyglot branch returns
# before the end of this script.
apply_dns_jail() {
[ -f /workspace/.devcontainer/dns-jail.sh ] || return 0
# Read the one line rather than sourcing: this runs on every boot AND every run-app, and
# with the jail off it must not execute the worker's .env as a side effect.
if [ "${RACCOON_DNS_JAIL:-0}" != "1" ] &&
! grep -qE '^[[:space:]]*(export[[:space:]]+)?RACCOON_DNS_JAIL[[:space:]]*=[[:space:]]*"?1"?[[:space:]]*(#.*)?$' \
/workspace/.env 2>/dev/null; then
return 0
fi
(
set -a
# shellcheck disable=SC1091
. /workspace/.env 2>/dev/null || true
set +a
bash /workspace/.devcontainer/dns-jail.sh
) || true
}
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null || true)
if [ -n "$IS_POLYGLOT" ]; then
if ! pg_ready; then
@@ -129,7 +106,6 @@ if [ -n "$IS_POLYGLOT" ]; then
os_up || echo "warning: opensearch did not come up within 180s (see /tmp/opensearch.log)" >&2
fi
fi
apply_dns_jail
return 0 2>/dev/null || exit 0
fi
@@ -244,4 +220,3 @@ case "$REPO_NAME" in
;;
esac
apply_dns_jail

View File

@@ -0,0 +1,88 @@
/** The task's holistic-rubric template. Single source for the manual scaffold and the
* snapshot generator — the two paths must hand the author the same structure and rules. */
export const HOLISTIC_RUBRIC_SCAFFOLD = `# Holistic Rubric — <task-slug>
The shared grading standard (\`task-shared/grading-standard.md\`, embedded in
\`tests/grader-system-prompt-consolidated.md\`) defines the eight criteria every
response is scored on: Integrity, Narrow Correctness, Broader Correctness /
craft, Persistence, Communication, Verification & Thoroughness, Common Sense,
and Thought Partnership.
This file is the task's holistic rubric. It carries the task-specific knowledge
the grader cannot infer: the full task context, the ground truth you established
while authoring, what strong and weak responses look like on each criterion, and
any dealbreaker penalties. This document must stand alone. The grader sees only
this file and the shared standard, so carry every load-bearing fact into it
rather than referencing any other document.
Replace each bracketed section. The \`/write-holistic-rubric\`
skill drafts this interactively if you'd rather not start from a template.
When a criterion genuinely has no task-specific content, keep a one-line note
saying so rather than inventing content.
## Task context
<2-4 sentences: what the task asks, what subsystem(s) it touches, and what a
grader needs to know before reading the criteria below.>
## Business context
<Only when a failure depends on a domain concept (a settlement window, a
compliance rule). Delete this section otherwise.>
## Ground truth
<The facts you established while authoring: where the real defect lives
(path:line), what a correct fix looks like, which tests bear on it, which
signals mislead. The grader trusts this section over its own reading.>
## Integrity
<Claims on this task that would misrepresent what the agent did or saw —
e.g. asserting a file says X after reading it say Y. Charge only on an
observable basis.>
## Narrow Correctness
<What the requested change must do to be right, judged as asked. Anchors a
working result must satisfy, checkable by path:line.>
## Broader Correctness / the craft of software engineering
<Craft expectations specific to this codebase: patterns to follow, tests to
add, places a shortcut would rot.>
## Persistence
<What "kept going appropriately" looks like here: the dead ends worth
exhausting, and where stopping to ask is the better call.>
## Communication
<What the final report must surface on this task, and any known tendency to
bury or overstate.>
## Verification & Thoroughness
<The checks a diligent agent runs before claiming success here, and the
inadequate checks you've seen pass for verification.>
## Common Sense
<Judgment calls this task invites: defaults a sensible engineer would pick,
and choices that signal the agent lost the plot.>
## Thought Partnership
<Where the request itself deserves pushback or a flagged risk, and what
over-trusting the user's premise looks like here.>
## Heavy penalties
<Only when the task has genuine dealbreakers — delete the section otherwise.
Phrase each qualitatively, naming its target — a criterion ("apply a heavy
penalty to **Verification & Thoroughness**"), the overall score, or both —
never a numeric magnitude, never points, never a cap or pinned score: the
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
Never describe how criteria combine into an overall score.>
`;

View File

@@ -24,6 +24,7 @@ import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
@@ -747,40 +748,23 @@ if (lastUserMessage) {
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
HOLISTIC RUBRIC — the file trials grade against. Run
/write-holistic-rubric
to draft it interactively, or point Claude Code at this file,
session-full.jsonl, and task-shared/grading-standard.md.
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
## What happened in the snapshot conversation
What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
## What this file contains
The eight-criterion Grading Standard
(task-shared/grading-standard.md, embedded in
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
Correctness, Broader Correctness / craft, Persistence, Communication,
Verification & Thoroughness, Common Sense, and Thought Partnership. This
file adds the task-specific knowledge the grader cannot infer: full task
context, the ground truth you established, what strong and weak responses
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
fraction subtractions with a named criterion target, never points, never
caps. The document must stand alone: the grader sees only it and the
shared standard.
Draft this file with /write-holistic-rubric, or point your agent at it,
session-full.jsonl, and task-shared/grading-standard.md. Delete this comment
when you are done.
-->
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
including the instructions above. -->
`;
${HOLISTIC_RUBRIC_SCAFFOLD}`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');

View File

@@ -1 +0,0 @@
/home/eric/workspaces/dataannotation/current-project/worker-toolkit-potion-polyglot/repos

View File

@@ -395,9 +395,13 @@ CTR_MARKER_DIR="/opt/raccoon-setup"
# Record <repo> as set up in THIS container. Best-effort: if the marker can't be written the
# only consequence is that setup runs again next time, and every step of it is idempotent.
_mark_ctr_setup() { mkdir -p "$CTR_MARKER_DIR" 2>/dev/null && : > "$CTR_MARKER_DIR/$1.done" 2>/dev/null || true; }
# A member set up but left without its dependencies, so the later arms don't claim otherwise.
_mark_ctr_nodeps() { mkdir -p "$CTR_MARKER_DIR" 2>/dev/null && : > "$CTR_MARKER_DIR/$1.nodeps" 2>/dev/null || true; }
_clear_ctr_nodeps() { rm -f "$CTR_MARKER_DIR/$1.nodeps" 2>/dev/null || true; }
_ctr_nodeps() { [ -f "$CTR_MARKER_DIR/$1.nodeps" ]; }
# First-use setup for a member repo: checkout its commit, install deps, prepare DB.
# The DNS jail (post-start.sh) blocks package registries, and the setup below installs
# The DNS jail (apply-dns-jail.sh) blocks package registries, and the setup below installs
# from them. Lift it for the install, then put it back — including on Ctrl-C, or the
# container would silently keep its network until the next start.
_DNSJAIL_LIFTED=""
@@ -431,6 +435,7 @@ _dnsjail_restore() {
[ -n "$(ls -A /tmp/.dnsjail/lifts 2>/dev/null)" ] && return 0
[ -f /tmp/.dnsjail/allow ] || return 0
sudo env DNSJAIL_ALLOW="$(cat /tmp/.dnsjail/allow)" \
DNSJAIL_ALLOW_EXTRA="$(cat /tmp/.dnsjail/allow-extra 2>/dev/null)" \
sh /workspace/.devcontainer/dns-jail-container.sh >/dev/null 2>&1 || true
}
@@ -601,17 +606,28 @@ setup_repo() {
# outright (Explore resolved pydantic 2.13.4 against a lock pinning 2.9.2, and the
# pinned strawberry cannot import on 2.13). Prefer the lock when there is one.
local vdir; vdir=$(_uv_venv_dir "$repo")
# --clear: uv >=0.12 won't create over an existing venv, so a failed first setup
# would wedge the member. setuptools 80.x is the last line shipping pkg_resources.
( cd "$dir" \
&& uv venv "$vdir" -p "$ver" -q \
&& uv venv --clear "$vdir" -p "$ver" -q \
&& . "$vdir/bin/activate" \
&& uv pip install -q 'setuptools==80.9.0' \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { if [ -f poetry.lock ] && command -v poetry >/dev/null 2>&1 \
&& POETRY_VIRTUALENVS_CREATE=false poetry install -q --no-interaction --no-root 2>/dev/null; then true; \
elif [ -f pyproject.toml ]; then uv pip install -q -e . || uv pip install -q -r requirements.txt 2>/dev/null || true; \
elif [ -f requirements.txt ]; then uv pip install -q -r requirements.txt; \
elif [ -f server/requirements.txt ]; then uv pip install -q -r server/requirements.txt; \
elif [ -f setup.py ]; then uv pip install -q -e .; else true; fi; } ) || return 1
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } ) || return 1
# Non-fatal: a manifest that cannot resolve here (a pin with no wheel for this
# architecture, say) leaves the member readable instead of failing the whole run.
if ( cd "$dir" && . "$vdir/bin/activate" \
&& { if [ -f poetry.lock ] && command -v poetry >/dev/null 2>&1 \
&& POETRY_VIRTUALENVS_CREATE=false poetry install -q --no-interaction --no-root 2>/dev/null; then true; \
elif [ -f pyproject.toml ]; then uv pip install -q -e . || uv pip install -q -r requirements.txt 2>/dev/null || true; \
elif [ -f requirements.txt ]; then uv pip install -q -r requirements.txt; \
elif [ -f server/requirements.txt ]; then uv pip install -q -r server/requirements.txt; \
elif [ -f setup.py ]; then uv pip install -q -e .; else true; fi; } ); then
_clear_ctr_nodeps "$repo"
else
_mark_ctr_nodeps "$repo"
printf " ${YELLOW}\xe2\x9a\xa0 %s: dependencies did not install${RESET} \xe2\x80\x94 explore-only in this container.\n ${GRAY}Reading the code, git and the editor still work; running the app or its tests will not.\n Usually a pin with no build for this machine's architecture. To retry: rm %s/%s.done${RESET}\n" "$repo" "$CTR_MARKER_DIR" "$repo"
fi
_mark_ctr_setup "$repo"; return 0
fi
_py_have "$ver" || { printf " ${GRAY}(Python %s not in this image; skipping deps \xe2\x80\x94 explore-only)${RESET}\n" "$ver"; _mark_ctr_setup "$repo"; return 0; }
@@ -630,10 +646,9 @@ setup_repo() {
# Every member Dockerfile sets this; without it poetry builds a .venv here
# that the trial image has no equivalent of.
poetry config virtualenvs.create false 2>/dev/null || true; \
# The git→path rewrite invalidates poetry.lock ("changed significantly");
# regenerate it before installing. Poetry 2.x `lock` preserves pins by
# default (the old `--no-update` flag was removed in 2.0).
poetry lock 2>/dev/null || true; \
# The git→path rewrite invalidates poetry.lock ("changed significantly"), so
# re-lock preserving pins: --no-update on poetry 1.x, the default on 2.x.
poetry lock --no-update 2>/dev/null || poetry lock 2>/dev/null || true; \
# --no-root: install deps only, not the project package itself. Some members'
# pyproject package name doesn't map to a folder poetry can find ("No file/folder
# found for package <x>"), which fails the whole install. The worker explores +
@@ -684,7 +699,12 @@ setup_repo() {
}
start_poly() {
local repo="${1:-}"; [ -z "$repo" ] && repo="$(_poly_default)"
local repo="${1:-}"
if [ -z "$repo" ]; then
repo="$(_poly_default)"
printf "${GRAY}no repo given — defaulting to '%s'. run-app <repo> picks another: %s${RESET}\n" \
"$repo" "$(_poly_repos | tr '\n' ' ')"
fi
if ! _poly_repos | grep -qx "$repo"; then
printf "${RED}unknown repo '%s'.${RESET} available: ${GRAY}%s${RESET}\n" "$repo" "$(_poly_repos | tr '\n' ' ')"
return 1
@@ -718,7 +738,8 @@ start_poly() {
return 0
fi
bash /workspace/.devcontainer/post-start.sh >/dev/null 2>&1 || true
setup_repo "$repo" || { printf "${RED}setup failed for %s${RESET} \xe2\x80\x94 ${GRAY}run-app --logs${RESET}\n" "$repo"; return 1; }
# Not --logs: setup runs before any app log exists, so its error is on screen, not in a file.
setup_repo "$repo" || { printf "${RED}setup failed for %s${RESET} \xe2\x80\x94 ${GRAY}see the error above${RESET}\n" "$repo"; return 1; }
local cmd=""
case "$kind" in
elixir)
@@ -812,6 +833,10 @@ start_poly() {
cmd="env $bootenv $craenv PATH=$nbin:\$PATH PORT=3000 BROWSER=none HOST=0.0.0.0 yarn $sc"
fi ;;
python)
if _ctr_nodeps "$repo"; then
printf " ${GRAY}%s: explore-only \xe2\x80\x94 its dependencies did not install, so its tests and scripts won't run here.${RESET}\n" "$repo"
return 0
fi
if [ -z "$startcmd" ]; then
printf " ${GRAY}%s: Python deps installed. No web server is wired \xe2\x80\x94 run its tests/scripts directly (e.g. pytest).${RESET}\n" "$repo"
return 0

View File

@@ -82,8 +82,23 @@ multi_agent_v2 = false
memories = false
external_agent_memory_import = false
"""
# A provider of our own, not the built-in `openai`: codex reserves built-in provider ids,
# and env_http_headers — the only place codex can be told to send the call-origin header —
# is a per-provider setting. base_url has to live in the table with it (a provider without
# one silently falls back to api.openai.com), so harness_refresh_config_keys refreshes
# [model_providers.*] keys as well as root ones.
container_config = """
openai_base_url = "${OPENAI_BASE_URL}"
model_provider = "llm-proxy"
[model_providers.llm-proxy]
name = "LLM proxy"
base_url = "${OPENAI_BASE_URL}"
# A custom provider reads its key from this env var and never from auth.json, so it
# names the one key .env actually carries. That makes the key live per launch rather
# than baked at container create — better than the auth-file path it replaces.
env_key = "ANTHROPIC_API_KEY"
wire_api = "responses"
env_http_headers = { "X-Surge-Client-Metadata" = "LLM_CALL_METADATA" }
"""
explore_config = """
[hooks]

View File

@@ -190,16 +190,39 @@ if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1
raise SystemExit(1)
text = os.path.expandvars(text)
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
# root. Every other table, [hooks] on the explore surface included, is left alone.
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
wanted = []
section = None
for line in text.splitlines():
if line.lstrip().startswith("["):
break
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
stripped = line.strip()
if stripped.startswith("["):
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
continue
if section is False:
continue
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
if m:
wanted.append((m.group(1), line.rstrip()))
wanted.append((section, m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
def section_path(header):
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
return tuple(header.strip("[]").split("."))
def lookup(doc, header, key):
"""The value a parsed config holds for a wanted key, or KeyError."""
node = doc
if header:
for part in section_path(header):
node = node[part]
return node[key]
mode = None
if os.path.exists(target):
try:
@@ -208,23 +231,52 @@ if os.path.exists(target):
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
# Everything from the first table header on belongs to a table. A key appended after
# one is reparented into it, so both the search and the insert stay above the line.
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
changed = False
for key, line in wanted:
# The quoted spelling is the same key: replacing it beats adding a duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
def span(header):
"""The line range a section owns, or None when the file has no such section.
Root is everything above the first table header: a key appended below one
would be reparented into it, so searches and inserts stay inside the span.
"""
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
if header is None:
return 0, (heads[0] if heads else len(lines))
at = next((i for i in heads if lines[i].strip() == header), None)
if at is None:
if root_end < len(lines) and lines[root_end].strip():
lines.insert(root_end, "")
lines.insert(root_end, line)
root_end += 1
changed = True
elif lines[at] != line:
lines[at] = line
return None
after = next((i for i in heads if i > at), len(lines))
return at + 1, after
# Grouped, root first, so a section this file lacks can be written whole.
grouped = {}
for header, key, line in wanted:
grouped.setdefault(header, []).append((key, line))
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
changed = False
for header in ordered:
if span(header) is None:
# A config written before this section existed. Write the whole table
# rather than leave a root key naming a provider that is not there.
if lines and lines[-1].strip():
lines.append("")
lines.append(header)
lines.extend(line for _, line in grouped[header])
changed = True
continue
for key, line in grouped[header]:
# Re-read the span: an insert for an earlier key moved it.
start, end = span(header)
# The quoted spelling is the same key: replace rather than duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
if at is None:
if end < len(lines) and lines[end].strip():
lines.insert(end, "")
lines.insert(end, line)
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
@@ -238,9 +290,22 @@ try:
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have reached the root.
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
raise SystemExit(1)
# parses while leaving the key unset. Require every key to have landed on the value the
# blob asks for, in its own section — skipping sections this file does not carry.
blob_doc = tomllib.loads(text)
for header, key, _ in wanted:
try:
expected = lookup(blob_doc, header, key)
except (KeyError, TypeError):
raise SystemExit(1)
try:
got = lookup(doc, header, key)
except (KeyError, TypeError):
if header is None:
raise SystemExit(1)
continue
if got != expected:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())

View File

@@ -34,4 +34,21 @@ _scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# No args is a valid call: refresh only, for a lifecycle hook.
[ "$#" -gt 0 ] || exit 0
# Outside the subshell, because these have to reach the exec'd command: codex now reads
# its key from $ANTHROPIC_API_KEY per request, and a non-login shell sourced neither
# .bashrc (the key, the call origin) nor the profile that puts the CLI on PATH.
# Failures stay swallowed — an unreadable .env must not stop the agent starting.
export PATH="$HOME/.local/bin:$PATH"
if [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
fi
if [ -f "$HOME/.raccoon-call-origin" ]; then
# shellcheck disable=SC1091
. "$HOME/.raccoon-call-origin" 2>/dev/null || true
fi
exec "$@"

View File

@@ -151,6 +151,19 @@ harness_install_launchers() {
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
set -euo pipefail
# A harness that reads its key from \$ENV per request needs .env in its environment,
# and only an interactive shell sources .bashrc — which is also where PATH picks up
# ~/.local/bin, where the CLI itself lives. Both are set here so a launch works the
# same either way, with the key .env holds right now.
export PATH="\$HOME/.local/bin:\$PATH"
if [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
. "\${RACCOON_ENV_FILE:-/workspace/.env}"
set +a
fi
if [ -f "\$HOME/.raccoon-call-origin" ]; then
. "\$HOME/.raccoon-call-origin"
fi
if [ -f "$note_src" ]; then
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
else

View File

@@ -1,298 +0,0 @@
{
"polyglot": true,
"repos": [
{
"repo": "lambda-cloudwatch-logs-to-loggly",
"defaultCommit": "f17e2d3",
"runtime": "node:14"
},
{
"repo": "lambda-potion-engagement",
"defaultCommit": "c64365b",
"runtime": "node:14"
},
{
"repo": "lambda-potion-schedular",
"defaultCommit": "0843570",
"runtime": "node:14"
},
{
"repo": "lambda-potion-transcription-scheduler",
"defaultCommit": "1a2e3d5",
"runtime": "node:14"
},
{
"repo": "lambda-video-processing",
"defaultCommit": "0e4a9b5",
"runtime": "node:18"
},
{
"repo": "microservice-dynamic-screen-recording",
"defaultCommit": "31e142b",
"runtime": "node:18"
},
{
"repo": "microservice-potion-voice",
"defaultCommit": "b65ca17",
"runtime": "node:14"
},
{
"repo": "potion-dynamic-screen-recording-lambda",
"defaultCommit": "57ed9e6",
"runtime": "node:14"
},
{
"repo": "potion-job-consumer",
"defaultCommit": "93f8a10",
"runtime": "node:18"
},
{
"repo": "potion-job-producer",
"defaultCommit": "04663d1",
"runtime": "node:18"
},
{
"repo": "potion-video-processing",
"defaultCommit": "59c6af9",
"runtime": "node:14"
},
{
"repo": "potion-voice",
"defaultCommit": "fcd8a9d",
"runtime": "node:14"
},
{
"repo": "potion-watcher",
"defaultCommit": "0e5973b",
"runtime": "node:18"
},
{
"repo": "potion-website-recording-handler",
"defaultCommit": "c58a9bb",
"runtime": "node:18"
},
{
"repo": "potion-app",
"defaultCommit": "6b4fee0c",
"runtime": "node:16",
"startCmd": "bash -c \"cp -n .env.client.development .env.local 2>/dev/null || true; export POTION_APP_ENV=local; [ -f .nuxt/store.js ] || npx nuxt build; node scripts/seed-dev-user.js || true; node server/index.js\"",
"setupCmd": "bash -c \"cp -n .env.client.development .env.local 2>/dev/null || true; export POTION_APP_ENV=local; [ -f .nuxt/store.js ] || npx nuxt build\""
},
{
"repo": "potion-custom-domain-app",
"defaultCommit": "01a7034",
"runtime": "none"
},
{
"repo": "potion-website",
"defaultCommit": "27995f8",
"runtime": "node:16"
},
{
"repo": "browser-extensions",
"defaultCommit": "b5e75d4",
"runtime": "node:18"
},
{
"repo": "gcp-application",
"defaultCommit": "469056f",
"runtime": "node:18"
},
{
"repo": "lambda-text-to-speech",
"defaultCommit": "99054ac",
"runtime": "node:18"
},
{
"repo": "potion-multi-dsr-watcher",
"defaultCommit": "c275d7f",
"runtime": "node:18",
"startCmd": "npx @google-cloud/functions-framework --target=potion-multi-dsr-watcher",
"bootEnv": "MONGODB_URI=mongodb://127.0.0.1:27017/potion_dev"
},
{
"repo": "potion-qa",
"defaultCommit": "3920e6c",
"runtime": "node:18"
},
{
"repo": "potion-snapshot-testing",
"defaultCommit": "a80eb8d",
"runtime": "node:18"
},
{
"repo": "potion-web",
"defaultCommit": "0a7e699",
"runtime": "node:18",
"startCmd": "npx nuxt dev --host 0.0.0.0 --port 3000",
"bootEnv": "POTION_APP_ENV=development BUGSNAG_FRONTEND_KEY=00000000000000000000000000000000 API_BASE_URL=http://localhost:4300 POTION_BASE_URL=http://localhost:4300"
},
{
"repo": "potion-analytics",
"defaultCommit": "43a7d23",
"runtime": "node:20"
},
{
"repo": "potion-api",
"defaultCommit": "5abe18f",
"runtime": "node:20"
},
{
"repo": "MODNet-with-training",
"defaultCommit": "dace325",
"runtime": "python:3.10"
},
{
"repo": "avds-cleaner",
"defaultCommit": "bd3a503",
"runtime": "python:3.10"
},
{
"repo": "avspeech",
"defaultCommit": "ca0f90d",
"runtime": "python:3.10"
},
{
"repo": "lambda-datadog-forwarder",
"defaultCommit": "a57ae74",
"runtime": "python:3.10"
},
{
"repo": "potion-ai",
"defaultCommit": "0e454d8",
"runtime": "python:3.10"
},
{
"repo": "potion-ai-cpu",
"defaultCommit": "ad61fa7",
"runtime": "python:3.10"
},
{
"repo": "potion-ai-gpu",
"defaultCommit": "8413d71",
"runtime": "python:3.10"
},
{
"repo": "potion-stitch",
"defaultCommit": "cfaed2f",
"runtime": "python:3.10"
},
{
"repo": "potion-tryon",
"defaultCommit": "b7da6a2",
"runtime": "python:3.10"
},
{
"repo": "potion-video-background-change",
"defaultCommit": "e6f2ea4",
"runtime": "python:3.10"
},
{
"repo": "potion-voice-dataset",
"defaultCommit": "f3d79d6",
"runtime": "python:3.10"
},
{
"repo": "potion-voice-utils",
"defaultCommit": "eadc48b",
"runtime": "python:3.10"
},
{
"repo": "sentence-split-service",
"defaultCommit": "32356d2",
"runtime": "python:3.10"
},
{
"repo": "urlbox-experiments",
"defaultCommit": "141fe18",
"runtime": "python:3.10"
},
{
"repo": "video-synth-api",
"defaultCommit": "167fcd7",
"runtime": "python:3.10"
},
{
"repo": "wav2lip-fa",
"defaultCommit": "8448ef0",
"runtime": "python:3.10"
},
{
"repo": "yeahsure-tryon",
"defaultCommit": "c8dee39",
"runtime": "python:3.10"
},
{
"repo": "gcp-infrastructure",
"defaultCommit": "a7dc5cc",
"runtime": "none"
},
{
"repo": "potion-ai-pretrained-models-infra",
"defaultCommit": "8a88770",
"runtime": "none"
},
{
"repo": "potion-app-infra",
"defaultCommit": "2107464",
"runtime": "none"
},
{
"repo": "potion-bastion",
"defaultCommit": "062af16",
"runtime": "none"
},
{
"repo": "potion-video-processing-devops",
"defaultCommit": "566286d",
"runtime": "none"
},
{
"repo": "elasticmq-container",
"defaultCommit": "de8acb5",
"runtime": "none"
},
{
"repo": "gcp-cloud-infrastructure",
"defaultCommit": "aa033c8",
"runtime": "none"
},
{
"repo": "potion-devops",
"defaultCommit": "84a4532",
"runtime": "none"
},
{
"repo": "potion-wp-site",
"defaultCommit": "cb71e3a",
"runtime": "none"
}
],
"defaultRepo": "potion-app",
"version": "2f696c53b4",
"blockedHosts": [
"sendpotion.com",
"www.sendpotion.com",
"app.sendpotion.com",
"staging.sendpotion.com",
"development.sendpotion.com",
"devleopment.sendpotion.com",
"meawww.sendpotion.com",
"blog.sendpotion.com",
"help.sendpotion.com",
"terms.sendpotion.com",
"pricing.sendpotion.com",
"videoassets.sendpotion.com",
"subtitleassets.sendpotion.com",
"audioassets.sendpotion.com",
"videoassets.staging.sendpotion.com",
"subtitleassets.staging.sendpotion.com",
"audioassets.staging.sendpotion.com"
],
"explorePorts": {
"clientHost": 4300,
"serverHost": null,
"livereloadHost": null,
"corpusHost": null
}
}