Project-2 baseline
This commit is contained in:
@@ -5,6 +5,7 @@ FROM node:16-bookworm
|
||||
|
||||
# System deps for this member's runtime + services.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ffmpeg \
|
||||
sudo \
|
||||
jq \
|
||||
ca-certificates \
|
||||
|
||||
@@ -10,6 +10,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV NODE_OPTIONS=--openssl-legacy-provider
|
||||
|
||||
# --- Python >=3.10 for the reduced-toolset agent's str_replace_editor ---
|
||||
# str_replace_editor (shebang `python3`) uses dataclass(kw_only=True) → it requires Python
|
||||
# >=3.10. The era-matched base ships Debian's older system python (ruby:3.2.1/3.1.2 → 3.9,
|
||||
@@ -94,15 +96,16 @@ RUN git init && \
|
||||
git add -A && \
|
||||
git commit -m "initial" --quiet
|
||||
|
||||
# Install JS deps via npm (no yarn.lock committed).
|
||||
RUN cd chrome/recorder && npm install --no-audit --no-fund
|
||||
# Install JS deps from the committed yarn.lock; fall back to a plain install if it drifted.
|
||||
RUN (cd chrome/recorder && yarn install --frozen-lockfile --network-timeout 600000) \
|
||||
|| (cd chrome/recorder && yarn install --network-timeout 600000)
|
||||
|
||||
# Fold the env-prep above into the baseline commit: tests/test.sh captures the agent's
|
||||
# work as the diff against it, so uncommitted setup edits ship as the agent's own.
|
||||
RUN git add -A && git commit --amend --no-edit --quiet
|
||||
|
||||
# Fail loudly if any load-bearing tool is missing.
|
||||
RUN for t in node npm claude; do \
|
||||
RUN for t in node yarn claude; do \
|
||||
command -v "$t" >/dev/null 2>&1 || { echo "FATAL: required tool '$t' missing from image" >&2; exit 1; }; \
|
||||
done; \
|
||||
echo "toolchain OK"
|
||||
|
||||
@@ -5,6 +5,7 @@ FROM node:20-bookworm
|
||||
|
||||
# System deps for this member's runtime + services.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ffmpeg \
|
||||
sudo \
|
||||
jq \
|
||||
ca-certificates \
|
||||
|
||||
@@ -5,6 +5,7 @@ FROM node:16-bookworm
|
||||
|
||||
# System deps for this member's runtime + services.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ffmpeg \
|
||||
sudo \
|
||||
jq \
|
||||
ca-certificates \
|
||||
|
||||
@@ -8,6 +8,7 @@ RUN printf 'deb http://archive.debian.org/debian bullseye main\ndeb http://archi
|
||||
&& printf 'Acquire::Check-Valid-Until "false";\nAcquire::Retries "5";\n' > /etc/apt/apt.conf.d/99no-check-valid-until \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
ffmpeg \
|
||||
sudo \
|
||||
jq \
|
||||
ca-certificates \
|
||||
|
||||
@@ -11,3 +11,5 @@ This directory contains the canonical copies of the files every Harbor task shar
|
||||
| `grader-system-prompt-consolidated.md` | `<task>/tests/grader-system-prompt-consolidated.md` | The grader system prompt. It embeds the Grading Standard; task-specific guidance is separate. |
|
||||
| `render-grade-consolidated.py` | `<task>/tests/render-grade-consolidated.py` | Renders the grader's grade.json into reward.txt and grade.md |
|
||||
| `grading-standard.md` | *(read in place)* | The Grading Standard: the eight criteria the grader scores |
|
||||
|
||||
`codex-grader.py` is copied alongside `test.sh` to run Codex and translate its response into the existing grade/result files.
|
||||
|
||||
110
worker-toolkit-potion-polyglot/task-shared/codex-grader.py
Normal file
110
worker-toolkit-potion-polyglot/task-shared/codex-grader.py
Normal file
@@ -0,0 +1,110 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the shared Codex CLI and emit the grade/session files test.sh already uses."""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
|
||||
def command(args, env, output):
|
||||
base = env.get('OPENAI_BASE_URL', '').strip().rstrip('/')
|
||||
if not base:
|
||||
anthropic = env.get('ANTHROPIC_BASE_URL', '').strip().rstrip('/')
|
||||
if not anthropic.endswith('/anthropic'):
|
||||
raise ValueError('Codex grading needs OPENAI_BASE_URL or the existing ANTHROPIC_BASE_URL proxy route')
|
||||
base = anthropic[:-len('/anthropic')] + '/openai/v1'
|
||||
key = (env.get('OPENAI_API_KEY') or env.get('ANTHROPIC_API_KEY') or '').strip()
|
||||
if not key:
|
||||
raise ValueError('Codex grading needs the runtime proxy API key')
|
||||
metadata = '{"origin":"harbor-grading"}'
|
||||
for header in env.get('ANTHROPIC_CUSTOM_HEADERS', '').splitlines():
|
||||
name, _, value = header.partition(':')
|
||||
if name.lower() == 'x-surge-client-metadata':
|
||||
metadata = value.strip()
|
||||
settings = {
|
||||
'model_provider': 'grader-proxy',
|
||||
'model_reasoning_effort': args.effort,
|
||||
'model_providers.grader-proxy.name': 'Grader proxy',
|
||||
'model_providers.grader-proxy.base_url': base,
|
||||
'model_providers.grader-proxy.env_key': 'OPENAI_API_KEY',
|
||||
'model_providers.grader-proxy.wire_api': 'responses',
|
||||
'model_providers.grader-proxy.http_headers.X-Surge-Client-Metadata': metadata,
|
||||
# The task container supplies the execution environment, as for Claude.
|
||||
'approval_policy': 'never', 'sandbox_mode': 'danger-full-access',
|
||||
}
|
||||
if args.one_shot:
|
||||
# Disabling shell alone leaves apply_patch available in Codex 0.158.0.
|
||||
catalog = output.parent / 'models.json'
|
||||
catalog.write_text(json.dumps({'models': [{
|
||||
'slug': args.model, 'display_name': args.model, 'description': 'Tool-free grading',
|
||||
'supported_reasoning_levels': [], 'shell_type': 'disabled',
|
||||
'visibility': 'hide', 'supported_in_api': True, 'priority': 0,
|
||||
'support_verbosity': False, 'apply_patch_tool_type': None,
|
||||
'truncation_policy': {'mode': 'tokens', 'limit': 10000},
|
||||
'experimental_supported_tools': [], 'tool_mode': 'direct',
|
||||
'model_messages': {'instructions_template': 'Grade only the supplied evidence. You have no tools.'},
|
||||
}]}))
|
||||
settings.update({'model_catalog_json': str(catalog), 'features.shell_tool': False,
|
||||
'features.view_image': False, 'web_search': 'disabled',
|
||||
'tools.update_plan.enabled': False, 'tools.experimental_request_user_input.enabled': False,
|
||||
'agents.enabled': False, 'features.goals': False})
|
||||
cmd = ['codex', 'exec', '--json', '--skip-git-repo-check', '--ignore-user-config',
|
||||
'--ignore-rules', '--model', args.model, '--output-last-message', str(output)]
|
||||
for name, value in settings.items():
|
||||
cmd += ['-c', name + '=' + json.dumps(value)]
|
||||
if args.resume:
|
||||
cmd += ['resume', args.resume]
|
||||
return cmd + ['-'], dict(env, OPENAI_API_KEY=key)
|
||||
|
||||
|
||||
def run(args, prompt, env):
|
||||
grade = Path(args.grade)
|
||||
grade.unlink(missing_ok=True)
|
||||
version = subprocess.run(['codex', '--version'], env=env, capture_output=True, text=True, check=True).stdout
|
||||
match = re.search(r'\b(\d+)\.(\d+)\.(\d+)([^\s]*)', version)
|
||||
if not match or match[4] or tuple(map(int, match.group(1, 2, 3))) < (0, 158, 0):
|
||||
raise ValueError('Codex grading requires stable CLI 0.158.0 or newer; rebuild the task image')
|
||||
if not args.one_shot:
|
||||
prompt += '\n\nReturn the COMPLETE grade JSON as your final message, without markdown fences, instead of writing the grade file. The verifier saves and validates it. Use shell reads to inspect the evidence and view_image for screenshots.\n'
|
||||
with tempfile.TemporaryDirectory(prefix='codex-grade-') as directory:
|
||||
output = Path(directory) / 'grade.txt'
|
||||
cmd, child = command(args, env, output)
|
||||
process = subprocess.run(cmd, input=prompt, stdout=subprocess.PIPE, text=True, env=child)
|
||||
# Codex may have followed the original file-writing instruction before failing.
|
||||
grade.unlink(missing_ok=True)
|
||||
if process.returncode:
|
||||
raise ValueError(f'Codex exited with status {process.returncode}')
|
||||
events = [json.loads(line) for line in process.stdout.splitlines() if line.strip()]
|
||||
session = next((e.get('thread_id') for e in events if e.get('type') == 'thread.started'), None)
|
||||
completed = [e for e in events if e.get('type') == 'turn.completed']
|
||||
if not session or not completed or any(e.get('type') in {'error', 'turn.failed'} for e in events):
|
||||
raise ValueError('Codex did not complete a grading turn; ' + process.stdout)
|
||||
if args.resume and session != args.resume:
|
||||
raise ValueError('Codex resumed a different session')
|
||||
grade.write_text(output.read_text())
|
||||
# Preserve the result shape the existing repair loop reads.
|
||||
print(json.dumps({'session_id': session, 'usage': completed[-1].get('usage', {})}))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--model', required=True)
|
||||
parser.add_argument('--effort', required=True)
|
||||
parser.add_argument('--grade', required=True)
|
||||
parser.add_argument('--resume', default='')
|
||||
parser.add_argument('--one-shot', action='store_true')
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
run(args, sys.stdin.read(), dict(os.environ))
|
||||
except (OSError, ValueError, subprocess.SubprocessError) as error:
|
||||
print(f'ERROR: {error}', file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -76,7 +76,10 @@ dnsjail_apply() {
|
||||
drop_ours
|
||||
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
|
||||
# one would rather than an answer this resolver decided to keep.
|
||||
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
|
||||
# -u root: dnsmasq 2.80 (buster and older bases) drops to "nobody" and calls capset to
|
||||
# retain CAP_NET_ADMIN, which docker's default cap set does not grant -- so it exits and
|
||||
# the jail fails open on every such image.
|
||||
dnsmasq -u root --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
|
||||
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
|
||||
>/dev/null 2>>"$STATE/dnsmasq.err" || true
|
||||
fi
|
||||
|
||||
@@ -18,9 +18,11 @@ the grader agent score each atomic rubric criterion independently and write
|
||||
certain_dealbreaker=5 (Critical), possible_dealbreaker=2 (Major),
|
||||
unlikely_dealbreaker=1 (Minor); dodged_bullet criteria are weighted by
|
||||
their severity like every other category. Criteria whose manifest
|
||||
category is extra_credit carry weight 1 and are included only when their
|
||||
value is > 0 (fulfilled extra credit joins the weighted mean; unfulfilled
|
||||
extra credit is excluded rather than penalized). A non-extra-credit
|
||||
category is extra_credit carry weight 1 scaled by how far they were
|
||||
fulfilled, and enter the mean at full value: a pass joins at weight 1, a
|
||||
partial at weight 0.5, a scalar score s at weight s, and unfulfilled extra
|
||||
credit is left out. Extra credit therefore only ever raises the reward. A
|
||||
non-extra-credit
|
||||
criterion with a null/missing severity falls back to
|
||||
unlikely_dealbreaker (weight 1) with a warning on stderr,
|
||||
4. rewrites rubric-grade.json in normalized form (generator stamp).
|
||||
@@ -49,7 +51,7 @@ import os
|
||||
import sys
|
||||
from typing import Any, Dict, List
|
||||
|
||||
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.0.0"
|
||||
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.1.0"
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
FORMS = ("trinary", "scalar")
|
||||
@@ -244,30 +246,41 @@ def aggregate(grade: Dict[str, Any], expected: List[Dict[str, Any]]) -> Dict[str
|
||||
"""Severity-weighted mean over criteria in cents.
|
||||
|
||||
reward_cents = round_half_up(sum(weight_i * cents_i) / sum(weight_i))
|
||||
over included criteria. extra_credit (weight 1) is included only when its
|
||||
value is > 0; every other criterion is always included at its severity
|
||||
weight.
|
||||
over included criteria. Every criterion other than extra_credit is always
|
||||
included at its severity weight. An extra_credit criterion with value > 0
|
||||
is included at full value (100 cents) with its weight scaled by its value,
|
||||
so a partial counts as a pass at half weight and extra credit can only
|
||||
raise the reward; one with value 0 is left out.
|
||||
|
||||
Sums are kept in hundredths of a weight unit so the arithmetic stays exact.
|
||||
"""
|
||||
weighted_cents = 0
|
||||
total_weight = 0
|
||||
weighted = 0 # sum of weight * cents, in hundredths of a weight unit
|
||||
total = 0 # sum of weight, in hundredths of a weight unit
|
||||
n_included = 0
|
||||
excluded_extra_credit = 0
|
||||
partial_extra_credit = 0
|
||||
for criterion in expected:
|
||||
entry = grade["by_id"][criterion["id"]]
|
||||
if criterion["category"] == "extra_credit" and entry["_cents"] == 0:
|
||||
excluded_extra_credit += 1
|
||||
cents = entry["_cents"]
|
||||
if criterion["category"] == "extra_credit":
|
||||
if cents == 0:
|
||||
excluded_extra_credit += 1
|
||||
continue
|
||||
if cents < 100:
|
||||
partial_extra_credit += 1
|
||||
n_included += 1
|
||||
weighted += criterion["weight"] * cents * 100
|
||||
total += criterion["weight"] * cents
|
||||
continue
|
||||
n_included += 1
|
||||
weighted_cents += criterion["weight"] * entry["_cents"]
|
||||
total_weight += criterion["weight"]
|
||||
if total_weight:
|
||||
reward_cents = _round_half_up(weighted_cents, total_weight)
|
||||
else:
|
||||
reward_cents = 0
|
||||
weighted += criterion["weight"] * cents * 100
|
||||
total += criterion["weight"] * 100
|
||||
reward_cents = _round_half_up(weighted, total) if total else 0
|
||||
return {
|
||||
"n_included": n_included,
|
||||
"n_excluded_extra_credit": excluded_extra_credit,
|
||||
"total_weight": total_weight,
|
||||
"n_partial_extra_credit": partial_extra_credit,
|
||||
"total_weight": total / 100.0,
|
||||
"reward_cents": reward_cents,
|
||||
}
|
||||
|
||||
@@ -292,6 +305,13 @@ def render_markdown(
|
||||
excluded,
|
||||
"on" if excluded == 1 else "a",
|
||||
)
|
||||
partial = agg["n_partial_extra_credit"]
|
||||
if partial:
|
||||
detail += "; %d partly fulfilled extra-credit criteri%s counted at %s" % (
|
||||
partial,
|
||||
"on" if partial == 1 else "a",
|
||||
"half weight" if form == "trinary" else "a weight equal to the score",
|
||||
)
|
||||
sections = ["Rubric score (%s): %s (%s)" % (form, _fmt(agg["reward_cents"]), detail)]
|
||||
|
||||
for criterion in expected:
|
||||
@@ -380,8 +400,16 @@ def main() -> int:
|
||||
f.write(normalized_json(grade, expected, args.form))
|
||||
|
||||
print(
|
||||
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d total_weight=%d"
|
||||
% (reward, args.form, agg["n_included"], agg["n_excluded_extra_credit"], agg["total_weight"])
|
||||
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d "
|
||||
"partial_extra_credit=%d total_weight=%g"
|
||||
% (
|
||||
reward,
|
||||
args.form,
|
||||
agg["n_included"],
|
||||
agg["n_excluded_extra_credit"],
|
||||
agg["n_partial_extra_credit"],
|
||||
agg["total_weight"],
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,5 +1,5 @@
|
||||
#!/bin/bash
|
||||
# Verifier: grades the agent's work with Claude Code.
|
||||
# Verifier: grades the agent's work with Codex or Claude Code.
|
||||
# GRADER_MODE: "agentic" (default) explores the workspace with tools; "one-shot"
|
||||
# grades from the transcript alone (no tools); "rubric-trinary" / "rubric-scalar"
|
||||
# grade agentically against the task's atomic rubric criteria (staged as
|
||||
@@ -15,7 +15,7 @@
|
||||
# tests/render-grade-consolidated.py. Rubric modes grade against their staged
|
||||
# assets instead.
|
||||
|
||||
TESTS_DIR="$(dirname "$0")"
|
||||
TESTS_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
GRADER_MODE="${GRADER_MODE:-agentic}"
|
||||
|
||||
RUBRIC_FORM=""
|
||||
@@ -53,7 +53,18 @@ RENDER_GRADE="$TESTS_DIR/render-grade-consolidated.py"
|
||||
|
||||
# Grader model + number of samples (graded GRADER_SAMPLES times and averaged to
|
||||
# reduce noise). Override with GRADER_MODEL=... / GRADER_SAMPLES=...
|
||||
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
|
||||
GRADER_HARNESS="${GRADER_HARNESS:-codex}"
|
||||
case "$GRADER_HARNESS" in
|
||||
codex)
|
||||
GRADER_MODEL="${GRADER_MODEL:-gpt-6-sol}"
|
||||
GRADER_REASONING_EFFORT="${GRADER_REASONING_EFFORT:-high}"
|
||||
CODEX_HOME=$(mktemp -d /tmp/codex-grader.XXXXXXXX) || exit 1
|
||||
export CODEX_HOME
|
||||
trap 'rm -rf "$CODEX_HOME"' EXIT
|
||||
;;
|
||||
claude) GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}" ;;
|
||||
*) echo "ERROR: GRADER_HARNESS must be codex or claude" >&2; exit 1 ;;
|
||||
esac
|
||||
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
|
||||
# The grader model's Claude Code floor. A task image installs Claude Code when it is first built
|
||||
@@ -62,7 +73,7 @@ GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
# reward file. Fail here with the reason instead. The check applies to the default grader
|
||||
# model; set GRADER_CLI_MIN to enforce a floor for another model.
|
||||
GRADER_CLI_MIN="${GRADER_CLI_MIN:-2.1.251}"
|
||||
if [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; then
|
||||
if [ "$GRADER_HARNESS" = claude ] && { [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; }; then
|
||||
_cli_ver="$(claude --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
|
||||
if [ -n "$_cli_ver" ] && [ "$(printf '%s\n%s\n' "$GRADER_CLI_MIN" "$_cli_ver" | sort -V | head -1)" != "$GRADER_CLI_MIN" ]; then
|
||||
echo "ERROR: this task image carries Claude Code $_cli_ver, but the grader model $GRADER_MODEL needs $GRADER_CLI_MIN or newer." >&2
|
||||
@@ -80,14 +91,19 @@ case "${GRADER_FAST_MODE:-}" in
|
||||
*) echo "ERROR: unknown GRADER_FAST_MODE '$GRADER_FAST_MODE' (expected true, 1, yes, false, 0, or empty)" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
if [ "$GRADER_HARNESS" = codex ] && [ "${#GRADER_FAST_FLAGS[@]}" -gt 0 ]; then
|
||||
echo "ERROR: --fast / GRADER_FAST_MODE is Claude-only" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p /logs/verifier
|
||||
|
||||
# GRADER-REGIME:BEGIN
|
||||
# What actually governed this grade. Nothing can recover a grading regime after
|
||||
# the fact, so it is captured here or not at all — and every step below tolerates
|
||||
# failure, because a provenance record must never be able to fail a grade.
|
||||
# Keep byte-identical to the copy in _smoke-test/_smoke-deletion-capture (asserted
|
||||
# by scripts/lib/grader-regime.test.ts, exercised by CI's regrade-smoke).
|
||||
# Keep byte-identical to _smoke-test/_smoke-deletion-capture in the parent repository.
|
||||
# Its scripts/lib/grader-regime.test.ts asserts this; CI's regrade-smoke exercises it.
|
||||
_regime_sha() { [ -f "${1:-}" ] && sha256sum "$1" 2>/dev/null | cut -d' ' -f1; }
|
||||
_regime_json() {
|
||||
if [ -z "${1:-}" ]; then printf 'null'; else
|
||||
@@ -117,6 +133,8 @@ cat 2>/dev/null > /logs/verifier/grader-regime.json <<REGIME_EOF || true
|
||||
"schema_version": 1,
|
||||
"captured_at": $(_regime_json "$(date -u +%Y-%m-%dT%H:%M:%SZ)"),
|
||||
"grader_mode": $(_regime_json "${GRADER_MODE:-}"),
|
||||
"grader_harness": $(_regime_json "$GRADER_HARNESS"),
|
||||
"grader_reasoning_effort": $(_regime_json "${GRADER_REASONING_EFFORT:-}"),
|
||||
"grader_model": $(_regime_json "${GRADER_MODEL:-}"),
|
||||
"grader_samples": $(_regime_json "$_regime_samples"),
|
||||
"grading_standard": $(_regime_json "$_regime_standard"),
|
||||
@@ -491,13 +509,20 @@ $DELIVERABLE
|
||||
$ONESHOT_FINAL
|
||||
REWARD: 0.XX (a number from 0.00 to 1.00)."
|
||||
|
||||
cd /tmp/files && claude \
|
||||
--bare \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools '' \
|
||||
-p "$ONESHOT_PROMPT" \
|
||||
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
|
||||
if [ "$GRADER_HARNESS" = codex ]; then
|
||||
printf '%s' "$ONESHOT_PROMPT" | python3 "$TESTS_DIR/codex-grader.py" \
|
||||
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" --one-shot \
|
||||
--grade /logs/verifier/grade.md \
|
||||
>/logs/verifier/grader-result.json 2>/logs/verifier/grader-stderr.log || exit 1
|
||||
else
|
||||
cd /tmp/files && claude \
|
||||
--bare \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools '' \
|
||||
-p "$ONESHOT_PROMPT" \
|
||||
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
|
||||
fi
|
||||
|
||||
# Anchor to the whole final REWARD line so "REWARD: 10" / "1.5" capture the
|
||||
# FULL number (10 / 1.5) and get rejected by the range check below, instead
|
||||
@@ -656,14 +681,12 @@ else
|
||||
GRADE_SCHEMA_REMINDER='{"schema_version": 1, "criteria": {"integrity": {"score": 0.00-1.00 or null, "rationale": "..."}, "narrow_correctness": {...}, "broader_correctness": {...}, "persistence": {...}, "communication": {...}, "verification_thoroughness": {...}, "common_sense": {...}, "thought_partnership": {...}}, "overall_penalties": [{"amount": 0.10, "reason": "..."}], "overall_score": 0.00-1.00, "closing": "optional"}'
|
||||
fi
|
||||
|
||||
echo "Launching Claude Code grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
|
||||
echo "Launching $GRADER_HARNESS grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
|
||||
|
||||
# Grade GRADER_SAMPLES times and ship the mean (averaging reduces re-grade noise).
|
||||
# Per-sample artifacts are kept as reward-N.txt / grade-N.md / grader-result-N.json;
|
||||
# the canonical grade.md etc. are copied from the sample closest to the mean. A
|
||||
# sample with no valid reward in [0,1] is skipped; need min(2, GRADER_SAMPLES) valid.
|
||||
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
|
||||
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
mkdir -p /tmp/outputs /logs/verifier
|
||||
[ -e /tmp/files ] || ln -sfn /workspace /tmp/files
|
||||
|
||||
@@ -697,14 +720,8 @@ for I in $(seq 1 "$GRADER_SAMPLES"); do
|
||||
rm -f /logs/verifier/reward.txt /logs/verifier/reward-correctness.txt \
|
||||
/logs/verifier/grade.md /logs/verifier/grade.json /logs/verifier/rubric-grade.json
|
||||
if [ -n "$RESUME_SID" ]; then
|
||||
# Repair turn: same session, same judgment, just fix the file.
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Bash Write \
|
||||
--output-format json \
|
||||
--resume "$RESUME_SID" \
|
||||
-p "The $GRADE_JSON_PATH you wrote could not be parsed:
|
||||
# Repair the same judgment using the original repair instructions.
|
||||
GRADER_INPUT="The $GRADE_JSON_PATH you wrote could not be parsed:
|
||||
|
||||
$LAST_ERR
|
||||
|
||||
@@ -721,19 +738,32 @@ Then confirm it parses:
|
||||
|
||||
python3 -c \"import json; json.load(open('$GRADE_JSON_PATH'))\"
|
||||
|
||||
Do not change any judgment. Do not shorten any rationale." \
|
||||
Do not change any judgment. Do not shorten any rationale."
|
||||
else
|
||||
GRADER_INPUT="$GRADER_PROMPT"
|
||||
fi
|
||||
printf '%s' "$GRADER_INPUT" > "$GRADER_PROMPT_PATH"
|
||||
if [ "$GRADER_HARNESS" = codex ]; then
|
||||
python3 "$TESTS_DIR/codex-grader.py" \
|
||||
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" \
|
||||
--grade "$GRADE_JSON_PATH" --resume "$RESUME_SID" \
|
||||
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
elif [ -n "$RESUME_SID" ]; then
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Bash Write \
|
||||
--output-format json --resume "$RESUME_SID" -p "$GRADER_INPUT" \
|
||||
>"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
else
|
||||
printf '%s' "$GRADER_PROMPT" > "$GRADER_PROMPT_PATH"
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Glob Grep Bash Write \
|
||||
--output-format json \
|
||||
-p \
|
||||
<"$GRADER_PROMPT_PATH" \
|
||||
>"/logs/verifier/grader-result-$I.json" \
|
||||
--output-format json -p \
|
||||
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
fi
|
||||
|
||||
|
||||
Reference in New Issue
Block a user