Project-2 baseline

This commit is contained in:
2026-10-04 21:19:23 -04:00
parent 0f04889edf
commit 213eb3c403
861 changed files with 1710 additions and 3363322 deletions

View File

@@ -1,8 +1,8 @@
{
"version": 1,
"stampedAt": "2026-09-15T00:07:53.720Z",
"stampedAt": "2026-09-29T20:06:24.427Z",
"files": {
"tests/test.sh": "9dba71df2a7e3e1c19f928835f5f963d24f46fc638efa0db6d4e77d6c748c8cc",
"tests/test.sh": "67d84a6d694718a777dfb9d79cb629206f769b92c67d66e19d9bb98bda168d04",
"tests/grader-system-prompt-consolidated.md": "032ce032728a8c0b2717478b929dbd7535e07c96ffe2e991097dd2c233543275"
}
}

View File

@@ -7,7 +7,7 @@ commit = "f89abccf"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "037cfcf94b"
toolkit_version = "1.0.0"
# Set true for a task about a UI: the trial gets Playwright + Chromium (`pw <script.js>`),
# and on claude the `Read` tool so the agent can view a screenshot it takes. Leave false
# when the point of the task is that something cannot be verified.
@@ -38,6 +38,7 @@ gpus = 0
allow_internet = true
[verifier.env]
GRADER_HARNESS = "codex"
ANTHROPIC_API_KEY = "${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "${ANTHROPIC_BASE_URL}"

View File

@@ -0,0 +1,110 @@
#!/usr/bin/env python3
"""Run the shared Codex CLI and emit the grade/session files test.sh already uses."""
import argparse
import json
import os
from pathlib import Path
import re
import subprocess
import sys
import tempfile
def command(args, env, output):
base = env.get('OPENAI_BASE_URL', '').strip().rstrip('/')
if not base:
anthropic = env.get('ANTHROPIC_BASE_URL', '').strip().rstrip('/')
if not anthropic.endswith('/anthropic'):
raise ValueError('Codex grading needs OPENAI_BASE_URL or the existing ANTHROPIC_BASE_URL proxy route')
base = anthropic[:-len('/anthropic')] + '/openai/v1'
key = (env.get('OPENAI_API_KEY') or env.get('ANTHROPIC_API_KEY') or '').strip()
if not key:
raise ValueError('Codex grading needs the runtime proxy API key')
metadata = '{"origin":"harbor-grading"}'
for header in env.get('ANTHROPIC_CUSTOM_HEADERS', '').splitlines():
name, _, value = header.partition(':')
if name.lower() == 'x-surge-client-metadata':
metadata = value.strip()
settings = {
'model_provider': 'grader-proxy',
'model_reasoning_effort': args.effort,
'model_providers.grader-proxy.name': 'Grader proxy',
'model_providers.grader-proxy.base_url': base,
'model_providers.grader-proxy.env_key': 'OPENAI_API_KEY',
'model_providers.grader-proxy.wire_api': 'responses',
'model_providers.grader-proxy.http_headers.X-Surge-Client-Metadata': metadata,
# The task container supplies the execution environment, as for Claude.
'approval_policy': 'never', 'sandbox_mode': 'danger-full-access',
}
if args.one_shot:
# Disabling shell alone leaves apply_patch available in Codex 0.158.0.
catalog = output.parent / 'models.json'
catalog.write_text(json.dumps({'models': [{
'slug': args.model, 'display_name': args.model, 'description': 'Tool-free grading',
'supported_reasoning_levels': [], 'shell_type': 'disabled',
'visibility': 'hide', 'supported_in_api': True, 'priority': 0,
'support_verbosity': False, 'apply_patch_tool_type': None,
'truncation_policy': {'mode': 'tokens', 'limit': 10000},
'experimental_supported_tools': [], 'tool_mode': 'direct',
'model_messages': {'instructions_template': 'Grade only the supplied evidence. You have no tools.'},
}]}))
settings.update({'model_catalog_json': str(catalog), 'features.shell_tool': False,
'features.view_image': False, 'web_search': 'disabled',
'tools.update_plan.enabled': False, 'tools.experimental_request_user_input.enabled': False,
'agents.enabled': False, 'features.goals': False})
cmd = ['codex', 'exec', '--json', '--skip-git-repo-check', '--ignore-user-config',
'--ignore-rules', '--model', args.model, '--output-last-message', str(output)]
for name, value in settings.items():
cmd += ['-c', name + '=' + json.dumps(value)]
if args.resume:
cmd += ['resume', args.resume]
return cmd + ['-'], dict(env, OPENAI_API_KEY=key)
def run(args, prompt, env):
grade = Path(args.grade)
grade.unlink(missing_ok=True)
version = subprocess.run(['codex', '--version'], env=env, capture_output=True, text=True, check=True).stdout
match = re.search(r'\b(\d+)\.(\d+)\.(\d+)([^\s]*)', version)
if not match or match[4] or tuple(map(int, match.group(1, 2, 3))) < (0, 158, 0):
raise ValueError('Codex grading requires stable CLI 0.158.0 or newer; rebuild the task image')
if not args.one_shot:
prompt += '\n\nReturn the COMPLETE grade JSON as your final message, without markdown fences, instead of writing the grade file. The verifier saves and validates it. Use shell reads to inspect the evidence and view_image for screenshots.\n'
with tempfile.TemporaryDirectory(prefix='codex-grade-') as directory:
output = Path(directory) / 'grade.txt'
cmd, child = command(args, env, output)
process = subprocess.run(cmd, input=prompt, stdout=subprocess.PIPE, text=True, env=child)
# Codex may have followed the original file-writing instruction before failing.
grade.unlink(missing_ok=True)
if process.returncode:
raise ValueError(f'Codex exited with status {process.returncode}')
events = [json.loads(line) for line in process.stdout.splitlines() if line.strip()]
session = next((e.get('thread_id') for e in events if e.get('type') == 'thread.started'), None)
completed = [e for e in events if e.get('type') == 'turn.completed']
if not session or not completed or any(e.get('type') in {'error', 'turn.failed'} for e in events):
raise ValueError('Codex did not complete a grading turn; ' + process.stdout)
if args.resume and session != args.resume:
raise ValueError('Codex resumed a different session')
grade.write_text(output.read_text())
# Preserve the result shape the existing repair loop reads.
print(json.dumps({'session_id': session, 'usage': completed[-1].get('usage', {})}))
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--model', required=True)
parser.add_argument('--effort', required=True)
parser.add_argument('--grade', required=True)
parser.add_argument('--resume', default='')
parser.add_argument('--one-shot', action='store_true')
args = parser.parse_args()
try:
run(args, sys.stdin.read(), dict(os.environ))
except (OSError, ValueError, subprocess.SubprocessError) as error:
print(f'ERROR: {error}', file=sys.stderr)
return 1
return 0
if __name__ == '__main__':
sys.exit(main())

View File

@@ -18,9 +18,11 @@ the grader agent score each atomic rubric criterion independently and write
certain_dealbreaker=5 (Critical), possible_dealbreaker=2 (Major),
unlikely_dealbreaker=1 (Minor); dodged_bullet criteria are weighted by
their severity like every other category. Criteria whose manifest
category is extra_credit carry weight 1 and are included only when their
value is > 0 (fulfilled extra credit joins the weighted mean; unfulfilled
extra credit is excluded rather than penalized). A non-extra-credit
category is extra_credit carry weight 1 scaled by how far they were
fulfilled, and enter the mean at full value: a pass joins at weight 1, a
partial at weight 0.5, a scalar score s at weight s, and unfulfilled extra
credit is left out. Extra credit therefore only ever raises the reward. A
non-extra-credit
criterion with a null/missing severity falls back to
unlikely_dealbreaker (weight 1) with a warning on stderr,
4. rewrites rubric-grade.json in normalized form (generator stamp).
@@ -49,7 +51,7 @@ import os
import sys
from typing import Any, Dict, List
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.0.0"
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.1.0"
SCHEMA_VERSION = 1
FORMS = ("trinary", "scalar")
@@ -244,30 +246,41 @@ def aggregate(grade: Dict[str, Any], expected: List[Dict[str, Any]]) -> Dict[str
"""Severity-weighted mean over criteria in cents.
reward_cents = round_half_up(sum(weight_i * cents_i) / sum(weight_i))
over included criteria. extra_credit (weight 1) is included only when its
value is > 0; every other criterion is always included at its severity
weight.
over included criteria. Every criterion other than extra_credit is always
included at its severity weight. An extra_credit criterion with value > 0
is included at full value (100 cents) with its weight scaled by its value,
so a partial counts as a pass at half weight and extra credit can only
raise the reward; one with value 0 is left out.
Sums are kept in hundredths of a weight unit so the arithmetic stays exact.
"""
weighted_cents = 0
total_weight = 0
weighted = 0 # sum of weight * cents, in hundredths of a weight unit
total = 0 # sum of weight, in hundredths of a weight unit
n_included = 0
excluded_extra_credit = 0
partial_extra_credit = 0
for criterion in expected:
entry = grade["by_id"][criterion["id"]]
if criterion["category"] == "extra_credit" and entry["_cents"] == 0:
excluded_extra_credit += 1
cents = entry["_cents"]
if criterion["category"] == "extra_credit":
if cents == 0:
excluded_extra_credit += 1
continue
if cents < 100:
partial_extra_credit += 1
n_included += 1
weighted += criterion["weight"] * cents * 100
total += criterion["weight"] * cents
continue
n_included += 1
weighted_cents += criterion["weight"] * entry["_cents"]
total_weight += criterion["weight"]
if total_weight:
reward_cents = _round_half_up(weighted_cents, total_weight)
else:
reward_cents = 0
weighted += criterion["weight"] * cents * 100
total += criterion["weight"] * 100
reward_cents = _round_half_up(weighted, total) if total else 0
return {
"n_included": n_included,
"n_excluded_extra_credit": excluded_extra_credit,
"total_weight": total_weight,
"n_partial_extra_credit": partial_extra_credit,
"total_weight": total / 100.0,
"reward_cents": reward_cents,
}
@@ -292,6 +305,13 @@ def render_markdown(
excluded,
"on" if excluded == 1 else "a",
)
partial = agg["n_partial_extra_credit"]
if partial:
detail += "; %d partly fulfilled extra-credit criteri%s counted at %s" % (
partial,
"on" if partial == 1 else "a",
"half weight" if form == "trinary" else "a weight equal to the score",
)
sections = ["Rubric score (%s): %s (%s)" % (form, _fmt(agg["reward_cents"]), detail)]
for criterion in expected:
@@ -380,8 +400,16 @@ def main() -> int:
f.write(normalized_json(grade, expected, args.form))
print(
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d total_weight=%d"
% (reward, args.form, agg["n_included"], agg["n_excluded_extra_credit"], agg["total_weight"])
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d "
"partial_extra_credit=%d total_weight=%g"
% (
reward,
args.form,
agg["n_included"],
agg["n_excluded_extra_credit"],
agg["n_partial_extra_credit"],
agg["total_weight"],
)
)
return 0

View File

@@ -1,5 +1,5 @@
#!/bin/bash
# Verifier: grades the agent's work with Claude Code.
# Verifier: grades the agent's work with Codex or Claude Code.
# GRADER_MODE: "agentic" (default) explores the workspace with tools; "one-shot"
# grades from the transcript alone (no tools); "rubric-trinary" / "rubric-scalar"
# grade agentically against the task's atomic rubric criteria (staged as
@@ -15,7 +15,7 @@
# tests/render-grade-consolidated.py. Rubric modes grade against their staged
# assets instead.
TESTS_DIR="$(dirname "$0")"
TESTS_DIR="$(cd "$(dirname "$0")" && pwd)"
GRADER_MODE="${GRADER_MODE:-agentic}"
RUBRIC_FORM=""
@@ -53,7 +53,18 @@ RENDER_GRADE="$TESTS_DIR/render-grade-consolidated.py"
# Grader model + number of samples (graded GRADER_SAMPLES times and averaged to
# reduce noise). Override with GRADER_MODEL=... / GRADER_SAMPLES=...
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
GRADER_HARNESS="${GRADER_HARNESS:-codex}"
case "$GRADER_HARNESS" in
codex)
GRADER_MODEL="${GRADER_MODEL:-gpt-6-sol}"
GRADER_REASONING_EFFORT="${GRADER_REASONING_EFFORT:-high}"
CODEX_HOME=$(mktemp -d /tmp/codex-grader.XXXXXXXX) || exit 1
export CODEX_HOME
trap 'rm -rf "$CODEX_HOME"' EXIT
;;
claude) GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}" ;;
*) echo "ERROR: GRADER_HARNESS must be codex or claude" >&2; exit 1 ;;
esac
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
# The grader model's Claude Code floor. A task image installs Claude Code when it is first built
@@ -62,7 +73,7 @@ GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
# reward file. Fail here with the reason instead. The check applies to the default grader
# model; set GRADER_CLI_MIN to enforce a floor for another model.
GRADER_CLI_MIN="${GRADER_CLI_MIN:-2.1.251}"
if [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; then
if [ "$GRADER_HARNESS" = claude ] && { [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; }; then
_cli_ver="$(claude --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
if [ -n "$_cli_ver" ] && [ "$(printf '%s\n%s\n' "$GRADER_CLI_MIN" "$_cli_ver" | sort -V | head -1)" != "$GRADER_CLI_MIN" ]; then
echo "ERROR: this task image carries Claude Code $_cli_ver, but the grader model $GRADER_MODEL needs $GRADER_CLI_MIN or newer." >&2
@@ -80,14 +91,19 @@ case "${GRADER_FAST_MODE:-}" in
*) echo "ERROR: unknown GRADER_FAST_MODE '$GRADER_FAST_MODE' (expected true, 1, yes, false, 0, or empty)" >&2; exit 1 ;;
esac
if [ "$GRADER_HARNESS" = codex ] && [ "${#GRADER_FAST_FLAGS[@]}" -gt 0 ]; then
echo "ERROR: --fast / GRADER_FAST_MODE is Claude-only" >&2
exit 1
fi
mkdir -p /logs/verifier
# GRADER-REGIME:BEGIN
# What actually governed this grade. Nothing can recover a grading regime after
# the fact, so it is captured here or not at all — and every step below tolerates
# failure, because a provenance record must never be able to fail a grade.
# Keep byte-identical to the copy in _smoke-test/_smoke-deletion-capture (asserted
# by scripts/lib/grader-regime.test.ts, exercised by CI's regrade-smoke).
# Keep byte-identical to _smoke-test/_smoke-deletion-capture in the parent repository.
# Its scripts/lib/grader-regime.test.ts asserts this; CI's regrade-smoke exercises it.
_regime_sha() { [ -f "${1:-}" ] && sha256sum "$1" 2>/dev/null | cut -d' ' -f1; }
_regime_json() {
if [ -z "${1:-}" ]; then printf 'null'; else
@@ -117,6 +133,8 @@ cat 2>/dev/null > /logs/verifier/grader-regime.json <<REGIME_EOF || true
"schema_version": 1,
"captured_at": $(_regime_json "$(date -u +%Y-%m-%dT%H:%M:%SZ)"),
"grader_mode": $(_regime_json "${GRADER_MODE:-}"),
"grader_harness": $(_regime_json "$GRADER_HARNESS"),
"grader_reasoning_effort": $(_regime_json "${GRADER_REASONING_EFFORT:-}"),
"grader_model": $(_regime_json "${GRADER_MODEL:-}"),
"grader_samples": $(_regime_json "$_regime_samples"),
"grading_standard": $(_regime_json "$_regime_standard"),
@@ -491,13 +509,20 @@ $DELIVERABLE
$ONESHOT_FINAL
REWARD: 0.XX (a number from 0.00 to 1.00)."
cd /tmp/files && claude \
--bare \
--model "$GRADER_MODEL" \
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
--allowedTools '' \
-p "$ONESHOT_PROMPT" \
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
if [ "$GRADER_HARNESS" = codex ]; then
printf '%s' "$ONESHOT_PROMPT" | python3 "$TESTS_DIR/codex-grader.py" \
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" --one-shot \
--grade /logs/verifier/grade.md \
>/logs/verifier/grader-result.json 2>/logs/verifier/grader-stderr.log || exit 1
else
cd /tmp/files && claude \
--bare \
--model "$GRADER_MODEL" \
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
--allowedTools '' \
-p "$ONESHOT_PROMPT" \
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
fi
# Anchor to the whole final REWARD line so "REWARD: 10" / "1.5" capture the
# FULL number (10 / 1.5) and get rejected by the range check below, instead
@@ -656,14 +681,12 @@ else
GRADE_SCHEMA_REMINDER='{"schema_version": 1, "criteria": {"integrity": {"score": 0.00-1.00 or null, "rationale": "..."}, "narrow_correctness": {...}, "broader_correctness": {...}, "persistence": {...}, "communication": {...}, "verification_thoroughness": {...}, "common_sense": {...}, "thought_partnership": {...}}, "overall_penalties": [{"amount": 0.10, "reason": "..."}], "overall_score": 0.00-1.00, "closing": "optional"}'
fi
echo "Launching Claude Code grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
echo "Launching $GRADER_HARNESS grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
# Grade GRADER_SAMPLES times and ship the mean (averaging reduces re-grade noise).
# Per-sample artifacts are kept as reward-N.txt / grade-N.md / grader-result-N.json;
# the canonical grade.md etc. are copied from the sample closest to the mean. A
# sample with no valid reward in [0,1] is skipped; need min(2, GRADER_SAMPLES) valid.
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
mkdir -p /tmp/outputs /logs/verifier
[ -e /tmp/files ] || ln -sfn /workspace /tmp/files
@@ -697,14 +720,8 @@ for I in $(seq 1 "$GRADER_SAMPLES"); do
rm -f /logs/verifier/reward.txt /logs/verifier/reward-correctness.txt \
/logs/verifier/grade.md /logs/verifier/grade.json /logs/verifier/rubric-grade.json
if [ -n "$RESUME_SID" ]; then
# Repair turn: same session, same judgment, just fix the file.
cd /tmp/files && claude \
--model "$GRADER_MODEL" \
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
--allowedTools Read Bash Write \
--output-format json \
--resume "$RESUME_SID" \
-p "The $GRADE_JSON_PATH you wrote could not be parsed:
# Repair the same judgment using the original repair instructions.
GRADER_INPUT="The $GRADE_JSON_PATH you wrote could not be parsed:
$LAST_ERR
@@ -721,19 +738,32 @@ Then confirm it parses:
python3 -c \"import json; json.load(open('$GRADE_JSON_PATH'))\"
Do not change any judgment. Do not shorten any rationale." \
Do not change any judgment. Do not shorten any rationale."
else
GRADER_INPUT="$GRADER_PROMPT"
fi
printf '%s' "$GRADER_INPUT" > "$GRADER_PROMPT_PATH"
if [ "$GRADER_HARNESS" = codex ]; then
python3 "$TESTS_DIR/codex-grader.py" \
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" \
--grade "$GRADE_JSON_PATH" --resume "$RESUME_SID" \
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
2>"/logs/verifier/grader-stderr-$I.log"
elif [ -n "$RESUME_SID" ]; then
cd /tmp/files && claude \
--model "$GRADER_MODEL" \
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
--allowedTools Read Bash Write \
--output-format json --resume "$RESUME_SID" -p "$GRADER_INPUT" \
>"/logs/verifier/grader-result-$I.json" \
2>"/logs/verifier/grader-stderr-$I.log"
else
printf '%s' "$GRADER_PROMPT" > "$GRADER_PROMPT_PATH"
cd /tmp/files && claude \
--model "$GRADER_MODEL" \
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
--allowedTools Read Glob Grep Bash Write \
--output-format json \
-p \
<"$GRADER_PROMPT_PATH" \
>"/logs/verifier/grader-result-$I.json" \
--output-format json -p \
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
2>"/logs/verifier/grader-stderr-$I.log"
fi