#!/bin/bash
# Run raccoon tasks via Harbor with standard defaults.
#
# Automatically detects snapshot-based tasks (those with environment/session.jsonl)
# and uses the snapshot agent adapter for session resume.
#
# Usage: scripts/harbor-run <task-dir> [extra harbor args...]
# Example: scripts/harbor-run harbor-tasks/my-task-slug
# Example: scripts/harbor-run harbor-tasks/my-task-slug -k 4 --force-build
#
# To change the model, use --model (consumed here). Passing harbor's own -m does NOT
# override: harbor's -m is repeatable and builds one agent per value, so `-m X` runs the
# registry default AND X — two trials.
#
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.

set -euo pipefail

SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"

# Source API key
if [ -f "$REPO_ROOT/.env" ]; then
    set -a
    source "$REPO_ROOT/.env"
    set +a
fi

if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
    . "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi

# Per-harness credentials (OPENAI_API_KEY and friends) are DERIVED from the proxy root in
# ANTHROPIC_BASE_URL — they are not in .env. The container-create derivation exported them
# into a process that has long since exited, and only the auth FILES it wrote survive, so a
# fresh shell has the key on disk but not in its environment. resolve_harness checks the
# environment, and the trial passes it through to the sandbox, so re-derive here.
# Quiet on purpose: if it does not work, resolve_harness refuses by name a second later.
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
    # shellcheck disable=SC1091
    HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
    harness_setup_credentials >/dev/null 2>&1 || true
fi

export ANTHROPIC_BASE_URL="${ANTHROPIC_BASE_URL:-}"

# When running inside a devcontainer, harbor computes absolute paths for
# Docker bind mounts. These paths must be HOST paths because docker compose
# talks to the host daemon via the shared socket. Switching CWD to the
# host-equivalent workspace path makes harbor resolve paths correctly.
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
    cd "$HOST_WORKSPACE"
fi

TASK_DIR="$1"
shift

# Harness selection. `--harness` / `--model` / `--check-model` are consumed here;
# everything else passes through to harbor untouched, so existing invocations keep
# working. Parsed with a loop rather than getopts because the remaining args are an
# opaque harbor passthrough that getopts would try to interpret.
HARNESS_ARGS=()
PASSTHROUGH=()
while [ $# -gt 0 ]; do
    case "$1" in
        --harness) HARNESS_ARGS+=(--harness "$2"); shift 2 ;;
        --harness=*) HARNESS_ARGS+=(--harness "${1#*=}"); shift ;;
        --model) HARNESS_ARGS+=(--model "$2"); shift 2 ;;
        --model=*) HARNESS_ARGS+=(--model "${1#*=}"); shift ;;
        --check-model) HARNESS_ARGS+=(--check-model); shift ;;
        *) PASSTHROUGH+=("$1"); shift ;;
    esac
done
set -- "${PASSTHROUGH[@]+"${PASSTHROUGH[@]}"}"

# Preflight: workspace must be populated before harbor tries to docker-build it.
# Without this, the Dockerfile's `COPY workspace/ .` fails with an opaque
# "failed to calculate checksum of ref ...: \"/workspace\": not found" buried
# several frames deep in harbor's asyncio + docker-compose traceback. Surface
# the real fix here instead.
WORKSPACE_DIR="$TASK_DIR/environment/workspace"
if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ]; then
    SLUG="$(basename "$TASK_DIR")"
    echo "Error: $WORKSPACE_DIR is missing or empty." >&2
    echo "Build it first:  bash scripts/build-workspace.sh $SLUG" >&2
    echo "(reads commit from $TASK_DIR/task.toml; applies environment/workspace.patch if present.)" >&2
    exit 1
fi

# Preflight: recompute the browser marker from task.toml.
#
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
# by leaving it alone.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
    grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
    BROWSER_OPTIN=1
fi
if [ -d "$TASK_DIR/environment" ]; then
    printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
fi

# Preflight: report — never block — on edits to toolkit-managed files.
#
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt.md ship from
# task-shared/ and decide how the trial runs and how the grade is produced, so an edit
# makes a task's runs hard to compare with the rest. Surface that here, before a trial
# burns agent time. It is advisory on purpose: an author who edited one did it to get
# unstuck, and refusing to run their trial punishes a misunderstanding. `|| true` also
# means a checker that can't run (a fresh unzip with no node_modules) never reads as an
# edit. The checker only exists in the worker toolkit; here the file is absent.
CHECK_INFRA="$REPO_ROOT/scripts/check-task-infra.ts"
if [ -f "$CHECK_INFRA" ]; then
    (cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi

# Preflight: warn — loudly, but never block — when the live workspace has
# changes that a rebuild from the pinned commit + workspace.patch would lose.
# Trials run against the live workspace, but the finalized task keeps only the
# rebuild inputs (the workspace/ dir is gitignored) and every downstream
# consumer rebuilds from them, so anything uncaptured silently vanishes after
# packaging.
# The helper sits next to this script in a packed toolkit and under the
# toolkit's static scripts in the internal repo layout.
for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \
               "$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do
    if [ -f "$WS_SYNC" ]; then
        bash "$WS_SYNC" "$TASK_DIR" || true
        break
    fi
done

# Agent + model selection, from scripts/harness-registry.toml via resolve_harness.
# The agent classes come from scripts/{snapshot,codex,gemini}_agent.py or
# harness_agents.py (hence the PYTHONPATH). The Claude variants reuse the claude
# binary baked into the task image instead of re-downloading it at agent-setup —
# stock claude-code's runtime download (~240 MB) races the 360s agent-setup timeout
# and loses on slow-egress hosts (AgentSetupTimeoutError). Tasks that ship a
# non-empty environment/session.jsonl additionally resume the staged session;
# single-turn tasks get the non-resuming class.
#
# resolve_harness exits non-zero (and prints why) when the selection could not
# produce a usable grade — an unknown/disabled harness, one that writes no ATIF
# trajectory, a missing credential, or a task needing resume on a harness that
# can't. Failing here costs a second; failing later costs the whole trial, and the
# resume case wouldn't fail at all, it would silently grade the wrong thing.
# _raccoon_python comes from lib/harness-credentials.sh, sourced above. Define a fallback
# only if that file was missing, so the error below is about the interpreter rather than an
# unbound function.
command -v _raccoon_python >/dev/null 2>&1 || _raccoon_python() { return 1; }

export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
RACCOON_PY=$(_raccoon_python) || {
    echo "harbor-run: ERROR — no python3.11+ with tomllib on PATH, so the harness registry" >&2
    echo "harbor-run:   cannot be read and the agent class cannot be resolved. Set" >&2
    echo "harbor-run:   RACCOON_PYTHON to an interpreter that has tomllib (3.11+)." >&2
    exit 1
}
RESOLVED="$("$RACCOON_PY" "$SCRIPT_DIR/resolve_harness.py" \
    --task-dir "$TASK_DIR" \
    ${HARNESS_ARGS[@]+"${HARNESS_ARGS[@]}"})" || exit 1
eval "$RESOLVED"
# --agent-import-path, not --agent: harbor 0.20 deprecates it but still accepts it, and it
# is the flag whose recorded shape (`agents[0].import_path`) every run on disk and every
# reader expects. --agent leaves import_path null and puts the class in `name`, which
# silently empties the agent field in published benchmark rows. Revisit if the pin moves.
AGENT_FLAGS="--agent-import-path $AGENT_IMPORT_PATH"
# Empty EFFORT_KWARG means "run the harness's native default config" — pass no
# effort kwarg at all rather than an empty one, which harbor would reject.
EFFORT_FLAGS=""
[ -n "$EFFORT_KWARG" ] && EFFORT_FLAGS="--ak $EFFORT_KWARG=$EFFORT_VALUE"

# Environment backend. An explicit HARBOR_ENV always wins (either direction).
# Otherwise the default is context-dependent:
#   - daytona for the internal repo: runs the trial in a cloud sandbox over
#     HTTP, so it needs no local docker daemon and works *inside* the primary
#     devcontainer. (HARBOR_ENV=docker uses the host docker daemon instead —
#     free and offline, but host-only; the devcontainer has no docker.sock.)
#   - docker for the worker toolkit: it's provisioned only for the local docker
#     backend (docker CLI + bind-mounted docker.sock, no DAYTONA_API_KEY), so a
#     daytona default would just error out. We detect a toolkit checkout by
#     toolkit.json at the repo root — a file the packaging step writes that the
#     internal repo never has. It lives in the bind-mounted workspace, not a
#     baked image layer, so this holds even when the container's HARBOR_ENV pin
#     is missing (e.g. a stale, pre-pin image).
if [ -n "${HARBOR_ENV:-}" ]; then
    ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
    ENV_TYPE="docker"
else
    ENV_TYPE="daytona"
fi

# --no-delete keeps the environment around after the trial for inspection.
# That's free for a local docker container, but a Daytona sandbox is *billed*
# while it exists — keeping it would leak a paid sandbox on every run. Harbor
# downloads the trial logs into harbor-jobs before teardown either way, so for
# daytona we let it delete the sandbox; for docker we keep the container.
DELETE_FLAGS="--no-delete"
[ "$ENV_TYPE" = "daytona" ] && DELETE_FLAGS=""

# Orphan resilience (daytona): harbor tears sandboxes down per-trial + via an
# atexit that only closes the client — neither runs on SIGTERM/SIGKILL/crash, so
# a killed run leaks STARTED sandboxes that hog the shared pool until (if ever)
# an account default reaps them. Tell Daytona to auto-stop an IDLE sandbox after
# 20 min (auto-delete on stop), so orphans self-clean however the process dies.
# Safe for live trials: a running agent/grader keeps the sandbox active.
AUTOSTOP_FLAGS=""
[ "$ENV_TYPE" = "daytona" ] && AUTOSTOP_FLAGS="--ek auto_stop_interval_mins=20"

if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
    echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
    echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
    exit 1
fi

# Launch-time input-checksum capture. Snapshot the task inputs BEFORE harbor
# starts, so the recorded hashes are what the agent actually ran against —
# an input edited between this run and copy-reference-run no longer records
# post-edit state and masks staleness. The capture is stamped into the trial
# dirs after the run (below); copy-reference-run prefers it over its weaker
# capture-at-copy fallback. Advisory end to end: a failure here never blocks
# the run.
#
# The post-run stamping watches the default `harbor-jobs` output dir. If the
# caller overrides the output dir via extra args, we can't know where the
# trials will land — skip stamping and say so, rather than silently stamping
# nothing (runs then fall back to copy-time capture in copy-reference-run).
STAMP_FILE=""
for arg in "$@"; do
    case "$arg" in
        -o|--output*)
            echo "Note: custom harbor output dir passed ($arg) — skipping launch-time input-checksum stamping; reference runs will fall back to copy-time capture." >&2
            STAMP_FILE="skip"
            break
            ;;
    esac
done
if [ "$STAMP_FILE" != "skip" ]; then
    STAMP_FILE="$(mktemp)"
    if ! npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" capture "$TASK_DIR" --out "$STAMP_FILE"; then
        echo "Warning: could not capture launch-time input checksums (staleness will be judged from copy-time capture instead)" >&2
        rm -f "$STAMP_FILE"
        STAMP_FILE=""
    fi
else
    STAMP_FILE=""
fi
JOBS_BEFORE="$(ls -1 harbor-jobs 2>/dev/null || true)"

# The model comes from the registry (already resolved above) and is always a
# CONCRETE id, never a shorthand alias. On the manual (non-snapshot) path the agent
# passes the model via ANTHROPIC_MODEL, where a shorthand is NOT alias-resolved, so
# the configured base-URL endpoint rejects it (400 "Invalid model: <shorthand>") and
# trials die on turn 1. Bump `default_model` in scripts/harness-registry.toml when a
# newer model ships.
#
# harbor runs as a child (this script used to `exec` it, but the post-run
# stamping needs to run after harbor exits), so forward TERM/INT: a `kill`
# aimed at this wrapper's PID must take harbor down with it, not orphan a
# running job (this repo has been bitten by zombie harbor coordinators
# before).
HARBOR_EXIT=0
HARBOR_SIGNALLED=""
harbor run \
    -p "$TASK_DIR" \
    $AGENT_FLAGS \
    -m "$MODEL" \
    -e "$ENV_TYPE" \
    $DELETE_FLAGS \
    $AUTOSTOP_FLAGS \
    --yes \
    -o harbor-jobs \
    $EFFORT_FLAGS \
    "$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
wait "$HARBOR_PID" || HARBOR_EXIT=$?
if [ -n "$HARBOR_SIGNALLED" ]; then
    # The first wait was interrupted by the trap; wait again so harbor's real
    # exit status (not the shell's signal status) is what we propagate.
    wait "$HARBOR_PID" || HARBOR_EXIT=$?
fi
trap - TERM INT

# Stamp the launch-time capture into this task's trial dirs in the job dir(s)
# this run created (the harbor-jobs entries that didn't exist before the
# run). Harbor names job dirs with a timestamp, so new entries are this run's
# output — plus, when several harbor-runs share a cwd, possibly a concurrent
# run's; `apply` is slug-scoped so another task's trials are never stamped
# with this task's inputs.
if [ -n "$STAMP_FILE" ]; then
    NEW_JOBS="$(comm -13 <(printf '%s\n' "$JOBS_BEFORE" | sort) <(ls -1 harbor-jobs 2>/dev/null | sort) | sed 's|^|harbor-jobs/|')"
    if [ -n "$NEW_JOBS" ]; then
        # shellcheck disable=SC2086 # job-dir names are timestamps, never spaced
        npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" apply "$STAMP_FILE" $NEW_JOBS || \
            echo "Warning: could not stamp trial dirs with launch-time input checksums" >&2
    fi
    rm -f "$STAMP_FILE"
fi

# Repeat the toolkit-managed-file notice AFTER the trial. The preflight copy is
# minutes of harbor output up the scrollback by now, which for a notice nothing
# enforces means nobody reads it. This one lands where the author is looking.
if [ -f "$CHECK_INFRA" ]; then
    (cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi

exit "$HARBOR_EXIT"
