Project-2 baseline
This commit is contained in:
@@ -0,0 +1,110 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the shared Codex CLI and emit the grade/session files test.sh already uses."""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
|
||||
def command(args, env, output):
|
||||
base = env.get('OPENAI_BASE_URL', '').strip().rstrip('/')
|
||||
if not base:
|
||||
anthropic = env.get('ANTHROPIC_BASE_URL', '').strip().rstrip('/')
|
||||
if not anthropic.endswith('/anthropic'):
|
||||
raise ValueError('Codex grading needs OPENAI_BASE_URL or the existing ANTHROPIC_BASE_URL proxy route')
|
||||
base = anthropic[:-len('/anthropic')] + '/openai/v1'
|
||||
key = (env.get('OPENAI_API_KEY') or env.get('ANTHROPIC_API_KEY') or '').strip()
|
||||
if not key:
|
||||
raise ValueError('Codex grading needs the runtime proxy API key')
|
||||
metadata = '{"origin":"harbor-grading"}'
|
||||
for header in env.get('ANTHROPIC_CUSTOM_HEADERS', '').splitlines():
|
||||
name, _, value = header.partition(':')
|
||||
if name.lower() == 'x-surge-client-metadata':
|
||||
metadata = value.strip()
|
||||
settings = {
|
||||
'model_provider': 'grader-proxy',
|
||||
'model_reasoning_effort': args.effort,
|
||||
'model_providers.grader-proxy.name': 'Grader proxy',
|
||||
'model_providers.grader-proxy.base_url': base,
|
||||
'model_providers.grader-proxy.env_key': 'OPENAI_API_KEY',
|
||||
'model_providers.grader-proxy.wire_api': 'responses',
|
||||
'model_providers.grader-proxy.http_headers.X-Surge-Client-Metadata': metadata,
|
||||
# The task container supplies the execution environment, as for Claude.
|
||||
'approval_policy': 'never', 'sandbox_mode': 'danger-full-access',
|
||||
}
|
||||
if args.one_shot:
|
||||
# Disabling shell alone leaves apply_patch available in Codex 0.158.0.
|
||||
catalog = output.parent / 'models.json'
|
||||
catalog.write_text(json.dumps({'models': [{
|
||||
'slug': args.model, 'display_name': args.model, 'description': 'Tool-free grading',
|
||||
'supported_reasoning_levels': [], 'shell_type': 'disabled',
|
||||
'visibility': 'hide', 'supported_in_api': True, 'priority': 0,
|
||||
'support_verbosity': False, 'apply_patch_tool_type': None,
|
||||
'truncation_policy': {'mode': 'tokens', 'limit': 10000},
|
||||
'experimental_supported_tools': [], 'tool_mode': 'direct',
|
||||
'model_messages': {'instructions_template': 'Grade only the supplied evidence. You have no tools.'},
|
||||
}]}))
|
||||
settings.update({'model_catalog_json': str(catalog), 'features.shell_tool': False,
|
||||
'features.view_image': False, 'web_search': 'disabled',
|
||||
'tools.update_plan.enabled': False, 'tools.experimental_request_user_input.enabled': False,
|
||||
'agents.enabled': False, 'features.goals': False})
|
||||
cmd = ['codex', 'exec', '--json', '--skip-git-repo-check', '--ignore-user-config',
|
||||
'--ignore-rules', '--model', args.model, '--output-last-message', str(output)]
|
||||
for name, value in settings.items():
|
||||
cmd += ['-c', name + '=' + json.dumps(value)]
|
||||
if args.resume:
|
||||
cmd += ['resume', args.resume]
|
||||
return cmd + ['-'], dict(env, OPENAI_API_KEY=key)
|
||||
|
||||
|
||||
def run(args, prompt, env):
|
||||
grade = Path(args.grade)
|
||||
grade.unlink(missing_ok=True)
|
||||
version = subprocess.run(['codex', '--version'], env=env, capture_output=True, text=True, check=True).stdout
|
||||
match = re.search(r'\b(\d+)\.(\d+)\.(\d+)([^\s]*)', version)
|
||||
if not match or match[4] or tuple(map(int, match.group(1, 2, 3))) < (0, 158, 0):
|
||||
raise ValueError('Codex grading requires stable CLI 0.158.0 or newer; rebuild the task image')
|
||||
if not args.one_shot:
|
||||
prompt += '\n\nReturn the COMPLETE grade JSON as your final message, without markdown fences, instead of writing the grade file. The verifier saves and validates it. Use shell reads to inspect the evidence and view_image for screenshots.\n'
|
||||
with tempfile.TemporaryDirectory(prefix='codex-grade-') as directory:
|
||||
output = Path(directory) / 'grade.txt'
|
||||
cmd, child = command(args, env, output)
|
||||
process = subprocess.run(cmd, input=prompt, stdout=subprocess.PIPE, text=True, env=child)
|
||||
# Codex may have followed the original file-writing instruction before failing.
|
||||
grade.unlink(missing_ok=True)
|
||||
if process.returncode:
|
||||
raise ValueError(f'Codex exited with status {process.returncode}')
|
||||
events = [json.loads(line) for line in process.stdout.splitlines() if line.strip()]
|
||||
session = next((e.get('thread_id') for e in events if e.get('type') == 'thread.started'), None)
|
||||
completed = [e for e in events if e.get('type') == 'turn.completed']
|
||||
if not session or not completed or any(e.get('type') in {'error', 'turn.failed'} for e in events):
|
||||
raise ValueError('Codex did not complete a grading turn; ' + process.stdout)
|
||||
if args.resume and session != args.resume:
|
||||
raise ValueError('Codex resumed a different session')
|
||||
grade.write_text(output.read_text())
|
||||
# Preserve the result shape the existing repair loop reads.
|
||||
print(json.dumps({'session_id': session, 'usage': completed[-1].get('usage', {})}))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--model', required=True)
|
||||
parser.add_argument('--effort', required=True)
|
||||
parser.add_argument('--grade', required=True)
|
||||
parser.add_argument('--resume', default='')
|
||||
parser.add_argument('--one-shot', action='store_true')
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
run(args, sys.stdin.read(), dict(os.environ))
|
||||
except (OSError, ValueError, subprocess.SubprocessError) as error:
|
||||
print(f'ERROR: {error}', file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -18,9 +18,11 @@ the grader agent score each atomic rubric criterion independently and write
|
||||
certain_dealbreaker=5 (Critical), possible_dealbreaker=2 (Major),
|
||||
unlikely_dealbreaker=1 (Minor); dodged_bullet criteria are weighted by
|
||||
their severity like every other category. Criteria whose manifest
|
||||
category is extra_credit carry weight 1 and are included only when their
|
||||
value is > 0 (fulfilled extra credit joins the weighted mean; unfulfilled
|
||||
extra credit is excluded rather than penalized). A non-extra-credit
|
||||
category is extra_credit carry weight 1 scaled by how far they were
|
||||
fulfilled, and enter the mean at full value: a pass joins at weight 1, a
|
||||
partial at weight 0.5, a scalar score s at weight s, and unfulfilled extra
|
||||
credit is left out. Extra credit therefore only ever raises the reward. A
|
||||
non-extra-credit
|
||||
criterion with a null/missing severity falls back to
|
||||
unlikely_dealbreaker (weight 1) with a warning on stderr,
|
||||
4. rewrites rubric-grade.json in normalized form (generator stamp).
|
||||
@@ -49,7 +51,7 @@ import os
|
||||
import sys
|
||||
from typing import Any, Dict, List
|
||||
|
||||
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.0.0"
|
||||
RENDER_RUBRIC_GRADE_VERSION = "render-rubric-grade/2.1.0"
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
FORMS = ("trinary", "scalar")
|
||||
@@ -244,30 +246,41 @@ def aggregate(grade: Dict[str, Any], expected: List[Dict[str, Any]]) -> Dict[str
|
||||
"""Severity-weighted mean over criteria in cents.
|
||||
|
||||
reward_cents = round_half_up(sum(weight_i * cents_i) / sum(weight_i))
|
||||
over included criteria. extra_credit (weight 1) is included only when its
|
||||
value is > 0; every other criterion is always included at its severity
|
||||
weight.
|
||||
over included criteria. Every criterion other than extra_credit is always
|
||||
included at its severity weight. An extra_credit criterion with value > 0
|
||||
is included at full value (100 cents) with its weight scaled by its value,
|
||||
so a partial counts as a pass at half weight and extra credit can only
|
||||
raise the reward; one with value 0 is left out.
|
||||
|
||||
Sums are kept in hundredths of a weight unit so the arithmetic stays exact.
|
||||
"""
|
||||
weighted_cents = 0
|
||||
total_weight = 0
|
||||
weighted = 0 # sum of weight * cents, in hundredths of a weight unit
|
||||
total = 0 # sum of weight, in hundredths of a weight unit
|
||||
n_included = 0
|
||||
excluded_extra_credit = 0
|
||||
partial_extra_credit = 0
|
||||
for criterion in expected:
|
||||
entry = grade["by_id"][criterion["id"]]
|
||||
if criterion["category"] == "extra_credit" and entry["_cents"] == 0:
|
||||
excluded_extra_credit += 1
|
||||
cents = entry["_cents"]
|
||||
if criterion["category"] == "extra_credit":
|
||||
if cents == 0:
|
||||
excluded_extra_credit += 1
|
||||
continue
|
||||
if cents < 100:
|
||||
partial_extra_credit += 1
|
||||
n_included += 1
|
||||
weighted += criterion["weight"] * cents * 100
|
||||
total += criterion["weight"] * cents
|
||||
continue
|
||||
n_included += 1
|
||||
weighted_cents += criterion["weight"] * entry["_cents"]
|
||||
total_weight += criterion["weight"]
|
||||
if total_weight:
|
||||
reward_cents = _round_half_up(weighted_cents, total_weight)
|
||||
else:
|
||||
reward_cents = 0
|
||||
weighted += criterion["weight"] * cents * 100
|
||||
total += criterion["weight"] * 100
|
||||
reward_cents = _round_half_up(weighted, total) if total else 0
|
||||
return {
|
||||
"n_included": n_included,
|
||||
"n_excluded_extra_credit": excluded_extra_credit,
|
||||
"total_weight": total_weight,
|
||||
"n_partial_extra_credit": partial_extra_credit,
|
||||
"total_weight": total / 100.0,
|
||||
"reward_cents": reward_cents,
|
||||
}
|
||||
|
||||
@@ -292,6 +305,13 @@ def render_markdown(
|
||||
excluded,
|
||||
"on" if excluded == 1 else "a",
|
||||
)
|
||||
partial = agg["n_partial_extra_credit"]
|
||||
if partial:
|
||||
detail += "; %d partly fulfilled extra-credit criteri%s counted at %s" % (
|
||||
partial,
|
||||
"on" if partial == 1 else "a",
|
||||
"half weight" if form == "trinary" else "a weight equal to the score",
|
||||
)
|
||||
sections = ["Rubric score (%s): %s (%s)" % (form, _fmt(agg["reward_cents"]), detail)]
|
||||
|
||||
for criterion in expected:
|
||||
@@ -380,8 +400,16 @@ def main() -> int:
|
||||
f.write(normalized_json(grade, expected, args.form))
|
||||
|
||||
print(
|
||||
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d total_weight=%d"
|
||||
% (reward, args.form, agg["n_included"], agg["n_excluded_extra_credit"], agg["total_weight"])
|
||||
"render-rubric-grade: ok reward=%s form=%s criteria=%d excluded_extra_credit=%d "
|
||||
"partial_extra_credit=%d total_weight=%g"
|
||||
% (
|
||||
reward,
|
||||
args.form,
|
||||
agg["n_included"],
|
||||
agg["n_excluded_extra_credit"],
|
||||
agg["n_partial_extra_credit"],
|
||||
agg["total_weight"],
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/bin/bash
|
||||
# Verifier: grades the agent's work with Claude Code.
|
||||
# Verifier: grades the agent's work with Codex or Claude Code.
|
||||
# GRADER_MODE: "agentic" (default) explores the workspace with tools; "one-shot"
|
||||
# grades from the transcript alone (no tools); "rubric-trinary" / "rubric-scalar"
|
||||
# grade agentically against the task's atomic rubric criteria (staged as
|
||||
@@ -15,7 +15,7 @@
|
||||
# tests/render-grade-consolidated.py. Rubric modes grade against their staged
|
||||
# assets instead.
|
||||
|
||||
TESTS_DIR="$(dirname "$0")"
|
||||
TESTS_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
GRADER_MODE="${GRADER_MODE:-agentic}"
|
||||
|
||||
RUBRIC_FORM=""
|
||||
@@ -53,7 +53,18 @@ RENDER_GRADE="$TESTS_DIR/render-grade-consolidated.py"
|
||||
|
||||
# Grader model + number of samples (graded GRADER_SAMPLES times and averaged to
|
||||
# reduce noise). Override with GRADER_MODEL=... / GRADER_SAMPLES=...
|
||||
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
|
||||
GRADER_HARNESS="${GRADER_HARNESS:-codex}"
|
||||
case "$GRADER_HARNESS" in
|
||||
codex)
|
||||
GRADER_MODEL="${GRADER_MODEL:-gpt-6-sol}"
|
||||
GRADER_REASONING_EFFORT="${GRADER_REASONING_EFFORT:-high}"
|
||||
CODEX_HOME=$(mktemp -d /tmp/codex-grader.XXXXXXXX) || exit 1
|
||||
export CODEX_HOME
|
||||
trap 'rm -rf "$CODEX_HOME"' EXIT
|
||||
;;
|
||||
claude) GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}" ;;
|
||||
*) echo "ERROR: GRADER_HARNESS must be codex or claude" >&2; exit 1 ;;
|
||||
esac
|
||||
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
|
||||
# The grader model's Claude Code floor. A task image installs Claude Code when it is first built
|
||||
@@ -62,7 +73,7 @@ GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
# reward file. Fail here with the reason instead. The check applies to the default grader
|
||||
# model; set GRADER_CLI_MIN to enforce a floor for another model.
|
||||
GRADER_CLI_MIN="${GRADER_CLI_MIN:-2.1.251}"
|
||||
if [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; then
|
||||
if [ "$GRADER_HARNESS" = claude ] && { [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; }; then
|
||||
_cli_ver="$(claude --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
|
||||
if [ -n "$_cli_ver" ] && [ "$(printf '%s\n%s\n' "$GRADER_CLI_MIN" "$_cli_ver" | sort -V | head -1)" != "$GRADER_CLI_MIN" ]; then
|
||||
echo "ERROR: this task image carries Claude Code $_cli_ver, but the grader model $GRADER_MODEL needs $GRADER_CLI_MIN or newer." >&2
|
||||
@@ -80,14 +91,19 @@ case "${GRADER_FAST_MODE:-}" in
|
||||
*) echo "ERROR: unknown GRADER_FAST_MODE '$GRADER_FAST_MODE' (expected true, 1, yes, false, 0, or empty)" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
if [ "$GRADER_HARNESS" = codex ] && [ "${#GRADER_FAST_FLAGS[@]}" -gt 0 ]; then
|
||||
echo "ERROR: --fast / GRADER_FAST_MODE is Claude-only" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p /logs/verifier
|
||||
|
||||
# GRADER-REGIME:BEGIN
|
||||
# What actually governed this grade. Nothing can recover a grading regime after
|
||||
# the fact, so it is captured here or not at all — and every step below tolerates
|
||||
# failure, because a provenance record must never be able to fail a grade.
|
||||
# Keep byte-identical to the copy in _smoke-test/_smoke-deletion-capture (asserted
|
||||
# by scripts/lib/grader-regime.test.ts, exercised by CI's regrade-smoke).
|
||||
# Keep byte-identical to _smoke-test/_smoke-deletion-capture in the parent repository.
|
||||
# Its scripts/lib/grader-regime.test.ts asserts this; CI's regrade-smoke exercises it.
|
||||
_regime_sha() { [ -f "${1:-}" ] && sha256sum "$1" 2>/dev/null | cut -d' ' -f1; }
|
||||
_regime_json() {
|
||||
if [ -z "${1:-}" ]; then printf 'null'; else
|
||||
@@ -117,6 +133,8 @@ cat 2>/dev/null > /logs/verifier/grader-regime.json <<REGIME_EOF || true
|
||||
"schema_version": 1,
|
||||
"captured_at": $(_regime_json "$(date -u +%Y-%m-%dT%H:%M:%SZ)"),
|
||||
"grader_mode": $(_regime_json "${GRADER_MODE:-}"),
|
||||
"grader_harness": $(_regime_json "$GRADER_HARNESS"),
|
||||
"grader_reasoning_effort": $(_regime_json "${GRADER_REASONING_EFFORT:-}"),
|
||||
"grader_model": $(_regime_json "${GRADER_MODEL:-}"),
|
||||
"grader_samples": $(_regime_json "$_regime_samples"),
|
||||
"grading_standard": $(_regime_json "$_regime_standard"),
|
||||
@@ -491,13 +509,20 @@ $DELIVERABLE
|
||||
$ONESHOT_FINAL
|
||||
REWARD: 0.XX (a number from 0.00 to 1.00)."
|
||||
|
||||
cd /tmp/files && claude \
|
||||
--bare \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools '' \
|
||||
-p "$ONESHOT_PROMPT" \
|
||||
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
|
||||
if [ "$GRADER_HARNESS" = codex ]; then
|
||||
printf '%s' "$ONESHOT_PROMPT" | python3 "$TESTS_DIR/codex-grader.py" \
|
||||
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" --one-shot \
|
||||
--grade /logs/verifier/grade.md \
|
||||
>/logs/verifier/grader-result.json 2>/logs/verifier/grader-stderr.log || exit 1
|
||||
else
|
||||
cd /tmp/files && claude \
|
||||
--bare \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools '' \
|
||||
-p "$ONESHOT_PROMPT" \
|
||||
> /logs/verifier/grade.md 2>/logs/verifier/grader-stderr.log
|
||||
fi
|
||||
|
||||
# Anchor to the whole final REWARD line so "REWARD: 10" / "1.5" capture the
|
||||
# FULL number (10 / 1.5) and get rejected by the range check below, instead
|
||||
@@ -656,14 +681,12 @@ else
|
||||
GRADE_SCHEMA_REMINDER='{"schema_version": 1, "criteria": {"integrity": {"score": 0.00-1.00 or null, "rationale": "..."}, "narrow_correctness": {...}, "broader_correctness": {...}, "persistence": {...}, "communication": {...}, "verification_thoroughness": {...}, "common_sense": {...}, "thought_partnership": {...}}, "overall_penalties": [{"amount": 0.10, "reason": "..."}], "overall_score": 0.00-1.00, "closing": "optional"}'
|
||||
fi
|
||||
|
||||
echo "Launching Claude Code grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
|
||||
echo "Launching $GRADER_HARNESS grader (requested model: $GRADER_MODEL, samples: $GRADER_SAMPLES)..."
|
||||
|
||||
# Grade GRADER_SAMPLES times and ship the mean (averaging reduces re-grade noise).
|
||||
# Per-sample artifacts are kept as reward-N.txt / grade-N.md / grader-result-N.json;
|
||||
# the canonical grade.md etc. are copied from the sample closest to the mean. A
|
||||
# sample with no valid reward in [0,1] is skipped; need min(2, GRADER_SAMPLES) valid.
|
||||
GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}"
|
||||
GRADER_SAMPLES="${GRADER_SAMPLES:-1}"
|
||||
mkdir -p /tmp/outputs /logs/verifier
|
||||
[ -e /tmp/files ] || ln -sfn /workspace /tmp/files
|
||||
|
||||
@@ -697,14 +720,8 @@ for I in $(seq 1 "$GRADER_SAMPLES"); do
|
||||
rm -f /logs/verifier/reward.txt /logs/verifier/reward-correctness.txt \
|
||||
/logs/verifier/grade.md /logs/verifier/grade.json /logs/verifier/rubric-grade.json
|
||||
if [ -n "$RESUME_SID" ]; then
|
||||
# Repair turn: same session, same judgment, just fix the file.
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Bash Write \
|
||||
--output-format json \
|
||||
--resume "$RESUME_SID" \
|
||||
-p "The $GRADE_JSON_PATH you wrote could not be parsed:
|
||||
# Repair the same judgment using the original repair instructions.
|
||||
GRADER_INPUT="The $GRADE_JSON_PATH you wrote could not be parsed:
|
||||
|
||||
$LAST_ERR
|
||||
|
||||
@@ -721,19 +738,32 @@ Then confirm it parses:
|
||||
|
||||
python3 -c \"import json; json.load(open('$GRADE_JSON_PATH'))\"
|
||||
|
||||
Do not change any judgment. Do not shorten any rationale." \
|
||||
Do not change any judgment. Do not shorten any rationale."
|
||||
else
|
||||
GRADER_INPUT="$GRADER_PROMPT"
|
||||
fi
|
||||
printf '%s' "$GRADER_INPUT" > "$GRADER_PROMPT_PATH"
|
||||
if [ "$GRADER_HARNESS" = codex ]; then
|
||||
python3 "$TESTS_DIR/codex-grader.py" \
|
||||
--model "$GRADER_MODEL" --effort "$GRADER_REASONING_EFFORT" \
|
||||
--grade "$GRADE_JSON_PATH" --resume "$RESUME_SID" \
|
||||
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
elif [ -n "$RESUME_SID" ]; then
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Bash Write \
|
||||
--output-format json --resume "$RESUME_SID" -p "$GRADER_INPUT" \
|
||||
>"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
else
|
||||
printf '%s' "$GRADER_PROMPT" > "$GRADER_PROMPT_PATH"
|
||||
cd /tmp/files && claude \
|
||||
--model "$GRADER_MODEL" \
|
||||
${GRADER_FAST_FLAGS[@]+"${GRADER_FAST_FLAGS[@]}"} \
|
||||
--allowedTools Read Glob Grep Bash Write \
|
||||
--output-format json \
|
||||
-p \
|
||||
<"$GRADER_PROMPT_PATH" \
|
||||
>"/logs/verifier/grader-result-$I.json" \
|
||||
--output-format json -p \
|
||||
<"$GRADER_PROMPT_PATH" >"/logs/verifier/grader-result-$I.json" \
|
||||
2>"/logs/verifier/grader-stderr-$I.log"
|
||||
fi
|
||||
|
||||
|
||||
Reference in New Issue
Block a user