Loaded up for the 3rd redo
Still on potion-voice
This commit is contained in:
201
worker-toolkit-potion-polyglot/scripts/replay_agent.py
Normal file
201
worker-toolkit-potion-polyglot/scripts/replay_agent.py
Normal file
@@ -0,0 +1,201 @@
|
||||
"""
|
||||
Harbor agent adapter that re-grades an existing reference run by replaying
|
||||
its captured workspace state — no model calls, no agent work.
|
||||
|
||||
Reads a `reference_run_dir` pointing at a `reference-runs/<id>/` directory
|
||||
captured during a prior real trial. Inside that dir:
|
||||
|
||||
agent/trajectory.json — the grader reads this via the symlink
|
||||
/tmp/outputs/task_transcript.txt → /logs/agent/trajectory.json
|
||||
agent-output/ — files the agent created or modified (captured by
|
||||
tests/test.sh after the agent ran)
|
||||
agent-output/_HARBOR_DELETIONS.txt
|
||||
— list of tracked files the agent deleted (one path
|
||||
per line). Empty/absent when nothing was deleted.
|
||||
|
||||
On run() (after the trial's docker env is up and /workspace has the base
|
||||
state from the Dockerfile's COPY workspace/), the adapter:
|
||||
|
||||
1. Uploads agent-output/ into /workspace — overlays the agent's surviving
|
||||
edits on top of the base workspace.
|
||||
2. Applies _HARBOR_DELETIONS.txt by `rm -f`'ing each listed path under
|
||||
/workspace. (The marker file itself was uploaded in step 1; it gets
|
||||
removed too so the verifier's re-capture doesn't pick it up as
|
||||
untracked content.)
|
||||
3. Uploads trajectory.json to /logs/agent/trajectory.json so the grader
|
||||
sees the same transcript it would have on the original run.
|
||||
|
||||
Verifier then runs as it would for any real trial — exact same code path,
|
||||
exact same artifacts, just with the agent phase replaced by a deterministic
|
||||
file-overlay. See scripts/harbor-regrade for the host-side wrapper.
|
||||
|
||||
Older reference runs captured before the deletion-capture line shipped
|
||||
(verifier-deletion-capture PR #219) won't have _HARBOR_DELETIONS.txt; the
|
||||
deletion-replay step is a no-op in that case. Deletions made in those runs
|
||||
remain lost.
|
||||
"""
|
||||
|
||||
from pathlib import Path, PurePosixPath
|
||||
|
||||
from harbor.agents.base import BaseAgent
|
||||
from harbor.environments.base import BaseEnvironment
|
||||
from harbor.models.agent.context import AgentContext
|
||||
from harbor.models.trial.paths import EnvironmentPaths
|
||||
|
||||
# Sibling module (shipped alongside in the worker toolkit; on PYTHONPATH via
|
||||
# harbor-regrade). Kept harbor-free so its logic stays unit-testable.
|
||||
from reference_run_capture import captured_mutating_tools
|
||||
|
||||
_WORKSPACE = PurePosixPath("/workspace")
|
||||
_DELETIONS_MARKER = "_HARBOR_DELETIONS.txt"
|
||||
|
||||
|
||||
class ReplayAgent(BaseAgent):
|
||||
"""Replays a captured reference run so the verifier can be re-graded
|
||||
without invoking the model again."""
|
||||
|
||||
SUPPORTS_WINDOWS: bool = False # paths below assume POSIX /workspace.
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
logs_dir,
|
||||
reference_run_dir: str,
|
||||
source_agent_import_path: str | None = None,
|
||||
source_model_name: str | None = None,
|
||||
**kwargs,
|
||||
):
|
||||
# `source_*` are provenance, not behaviour: harbor-regrade reads them off the
|
||||
# source run and passes them so THIS replay's result.json records which
|
||||
# harness and model produced the trajectory being graded. A replay reports
|
||||
# `replay_agent:ReplayAgent` with model_name null, and the source run is
|
||||
# usually deleted (a regrade is copied back over what it regraded), so a bare
|
||||
# reference_run_dir pointer does not survive as provenance.
|
||||
#
|
||||
# They must be accepted here rather than left in **kwargs: harbor records the
|
||||
# trial config's agent kwargs regardless of what the agent does with them, but
|
||||
# BaseAgent would reject the unknown keys and take every regrade down with it.
|
||||
self._source_agent_import_path = source_agent_import_path
|
||||
self._source_model_name = source_model_name
|
||||
super().__init__(logs_dir=logs_dir, **kwargs)
|
||||
ref = Path(reference_run_dir).expanduser().resolve()
|
||||
if not ref.is_dir():
|
||||
raise FileNotFoundError(f"reference_run_dir does not exist: {ref}")
|
||||
self._reference_run_dir = ref
|
||||
self._agent_output_dir = ref / "agent-output"
|
||||
self._trajectory_path = ref / "agent" / "trajectory.json"
|
||||
|
||||
@staticmethod
|
||||
def name() -> str:
|
||||
return "replay"
|
||||
|
||||
def version(self) -> str:
|
||||
return "1.0.0"
|
||||
|
||||
async def setup(self, environment: BaseEnvironment) -> None:
|
||||
# No installation needed; the verifier brings everything it requires.
|
||||
return
|
||||
|
||||
async def run(
|
||||
self,
|
||||
instruction: str,
|
||||
environment: BaseEnvironment,
|
||||
context: AgentContext,
|
||||
) -> None:
|
||||
# Advisory tasks — the agent only reads and answers in chat — make NO
|
||||
# workspace edits, so a faithful capture of one has an empty (or, in
|
||||
# older pipelines, absent) agent-output/. That is not a data gap: the
|
||||
# deliverable is the agent's final message, captured in
|
||||
# agent/trajectory.json, which the grader reads via
|
||||
# /tmp/outputs/task_transcript.txt. So overlay captured edits when
|
||||
# present; otherwise grade the base workspace + transcript, exactly
|
||||
# what the original advisory grading saw.
|
||||
if not self._agent_output_dir.is_dir():
|
||||
mutating = captured_mutating_tools(self._reference_run_dir)
|
||||
if mutating:
|
||||
# The agent edited files but they weren't captured — grading the
|
||||
# base workspace would silently score the wrong state. Refuse.
|
||||
raise FileNotFoundError(
|
||||
f"reference_run_dir {self._reference_run_dir} has no "
|
||||
f"agent-output/ but its captured transcript shows "
|
||||
f"file-mutating tool calls {sorted(mutating)}. The agent's "
|
||||
f"workspace edits were lost at capture time, so this run "
|
||||
f"cannot be faithfully re-graded — re-capture it."
|
||||
)
|
||||
self.logger.warning(
|
||||
"reference_run %s has no agent-output/ and made no "
|
||||
"file-mutating tool calls — treating it as an advisory run and "
|
||||
"grading the base workspace + captured transcript.",
|
||||
self._reference_run_dir,
|
||||
)
|
||||
# Skip the overlay/deletion steps; fall through to trajectory upload.
|
||||
await self._upload_trajectory(environment)
|
||||
return
|
||||
|
||||
# 1. Overlay captured agent edits onto the base /workspace.
|
||||
await environment.upload_dir(
|
||||
source_dir=str(self._agent_output_dir),
|
||||
target_dir=str(_WORKSPACE),
|
||||
)
|
||||
|
||||
# 2. Apply captured deletions, if present. Read the marker from the
|
||||
# host so we don't have to shell into the container to parse it,
|
||||
# then issue per-path rm's plus a final cleanup of the marker
|
||||
# itself (which was uploaded in step 1).
|
||||
host_marker = self._agent_output_dir / _DELETIONS_MARKER
|
||||
if host_marker.exists():
|
||||
deletion_paths = [
|
||||
line.strip()
|
||||
for line in host_marker.read_text().splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
for raw in deletion_paths:
|
||||
self._validate_relative_path(raw)
|
||||
await environment.exec(
|
||||
command=f'rm -f -- "/workspace/{raw}"',
|
||||
user="root",
|
||||
)
|
||||
await environment.exec(
|
||||
command=f'rm -f -- "/workspace/{_DELETIONS_MARKER}"',
|
||||
user="root",
|
||||
)
|
||||
|
||||
# 3. Materialize the captured trajectory at the path the grader's
|
||||
# test.sh symlinks to /tmp/outputs/task_transcript.txt.
|
||||
await self._upload_trajectory(environment)
|
||||
|
||||
async def _upload_trajectory(self, environment: BaseEnvironment) -> None:
|
||||
"""Upload agent/trajectory.json to the path the grader's test.sh
|
||||
symlinks to /tmp/outputs/task_transcript.txt. Shared by the normal
|
||||
(overlay) path and the advisory (no agent-output) path."""
|
||||
if self._trajectory_path.exists():
|
||||
env_paths = EnvironmentPaths.for_os(environment.os)
|
||||
await environment.upload_file(
|
||||
source_path=str(self._trajectory_path),
|
||||
target_path=str(env_paths.agent_dir / "trajectory.json"),
|
||||
)
|
||||
else:
|
||||
# The grader's test.sh reads /tmp/outputs/task_transcript.txt,
|
||||
# which symlinks to trajectory.json. Without the file the symlink
|
||||
# dangles and the grader sees an empty transcript — so the regrade
|
||||
# will look like the agent did nothing. Yell via harbor's own
|
||||
# logger (self.logger is a child of harbor.utils.logger) so the
|
||||
# warning lands in trial.log, not a stray "replay-agent" logger
|
||||
# nothing's wired to.
|
||||
self.logger.warning(
|
||||
"reference_run %s has no agent/trajectory.json — the grader "
|
||||
"will see an empty transcript. Investigate whether the source "
|
||||
"run was produced by an older harbor that didn't write the "
|
||||
"ATIF file (or by snapshot_agent before the multi-JSONL fix).",
|
||||
self._reference_run_dir,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_relative_path(raw: str) -> None:
|
||||
"""Guard against absolute paths and `..` traversal in the deletions
|
||||
manifest. The marker should only list paths *under* the workspace
|
||||
root; anything else is a captured-data integrity problem worth
|
||||
failing loudly on."""
|
||||
if not raw or raw.startswith("/"):
|
||||
raise ValueError(f"refusing absolute path in {_DELETIONS_MARKER}: {raw!r}")
|
||||
if ".." in PurePosixPath(raw).parts:
|
||||
raise ValueError(f"refusing `..` traversal in {_DELETIONS_MARKER}: {raw!r}")
|
||||
Reference in New Issue
Block a user