3 Commits

Author SHA1 Message Date
0d9a17695a Remove obsolete repos symlink 2026-10-05 14:05:46 -04:00
0f04889edf Restore staged files 2n shot 2026-09-29 20:17:47 -04:00
08a8379d5b Restore staged files 2026-09-29 20:16:33 -04:00
6 changed files with 27 additions and 278 deletions

View File

@@ -1 +0,0 @@
/home/eric/workspaces/dataannotation/current-project/worker-toolkit-potion-polyglot/repos

View File

@@ -1,33 +1,33 @@
{ {
"version": 1, "version": 1,
"generatedAt": "2026-09-28T22:38:24.037Z", "generatedAt": "2026-09-30T00:17:16.561Z",
"referenceRuns": [ "referenceRuns": [
{ {
"runId": "reward-0.3000-wNYgXoP", "runId": "reward-0.3500-42y7pDq",
"status": "fresh", "status": "fresh",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T18:35:34.158Z", "capturedAt": "2026-09-29T23:45:20.067Z",
"capturedBy": "copy" "capturedBy": "copy"
}, },
{ {
"runId": "reward-0.3700-J69VgLC", "runId": "reward-0.3700-fH3f28q",
"status": "fresh", "status": "fresh",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T18:39:33.253Z", "capturedAt": "2026-09-29T23:49:56.334Z",
"capturedBy": "copy" "capturedBy": "copy"
}, },
{ {
"runId": "reward-0.4500-h2zMRbJ", "runId": "reward-0.3800-EmMXDgM",
"status": "fresh", "status": "fresh",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T18:44:17.137Z", "capturedAt": "2026-09-29T23:40:48.353Z",
"capturedBy": "copy" "capturedBy": "copy"
}, },
{ {
"runId": "reward-0.5300-dHVmvQn", "runId": "reward-0.5900-aAV94ar",
"status": "fresh", "status": "fresh",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T18:48:49.336Z", "capturedAt": "2026-09-29T23:53:34.263Z",
"capturedBy": "copy" "capturedBy": "copy"
} }
], ],
@@ -37,7 +37,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T19:23:17.368Z", "capturedAt": "2026-09-29T23:29:34.541Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -45,7 +45,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T19:30:28.417Z", "capturedAt": "2026-09-29T23:29:35.047Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -53,7 +53,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T19:33:31.341Z", "capturedAt": "2026-09-29T23:29:35.550Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -61,7 +61,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T19:34:49.008Z", "capturedAt": "2026-09-29T23:29:36.065Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -69,7 +69,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T19:42:30.726Z", "capturedAt": "2026-09-29T23:29:36.560Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -77,7 +77,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T20:54:13.728Z", "capturedAt": "2026-09-29T23:29:37.064Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -85,7 +85,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T21:01:32.392Z", "capturedAt": "2026-09-29T23:29:37.563Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -93,7 +93,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T20:58:29.897Z", "capturedAt": "2026-09-29T23:29:38.050Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -101,7 +101,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T21:12:58.157Z", "capturedAt": "2026-09-29T23:29:38.556Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -109,7 +109,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:22:49.326Z", "capturedAt": "2026-09-29T23:29:39.071Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -117,7 +117,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:24:54.663Z", "capturedAt": "2026-09-29T23:29:39.577Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -125,7 +125,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:26:54.495Z", "capturedAt": "2026-09-29T23:29:40.069Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -133,7 +133,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:34:46.172Z", "capturedAt": "2026-09-29T23:29:40.557Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -141,7 +141,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T18:22:03.265Z", "capturedAt": "2026-09-29T23:29:41.063Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -149,7 +149,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:28:12.925Z", "capturedAt": "2026-09-29T23:29:41.573Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -157,7 +157,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:31:42.514Z", "capturedAt": "2026-09-29T23:29:42.083Z",
"capturedBy": "stamp" "capturedBy": "stamp"
}, },
{ {
@@ -165,7 +165,7 @@
"status": "fresh", "status": "fresh",
"method": "checksums", "method": "checksums",
"changed": [], "changed": [],
"capturedAt": "2026-09-28T22:33:08.497Z", "capturedAt": "2026-09-29T23:29:42.592Z",
"capturedBy": "stamp" "capturedBy": "stamp"
} }
] ]

View File

@@ -1,6 +0,0 @@
{
"created": [
"rubric-criteria.md",
"rubric-criteria.json"
]
}

View File

@@ -1,143 +0,0 @@
{
"task": "mishandled_pro_v2",
"criteria": [
{
"id": "pinpoints-flat-payload-crash",
"category": "primary_intent",
"severity": "certain_dealbreaker",
"dimensions": [
"Narrow Correctness"
]
},
{
"id": "supports-both-payload-envelopes",
"category": "primary_intent",
"severity": "certain_dealbreaker",
"dimensions": [
"Narrow Correctness"
]
},
{
"id": "audits-pro-v2-repository-state",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "surfaces-producer-contract-gap",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Thought Partnership",
"Communication"
]
},
{
"id": "confines-scope-to-transport-boundary",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Broader Correctness / the craft of software engineering"
]
},
{
"id": "avoids-ungrounded-tier-infrastructure",
"category": "dodged_bullet",
"severity": "crux",
"dimensions": [
"Thought Partnership"
]
},
{
"id": "limits-payload-normalization-to-evidenced-shapes",
"category": "primary_intent",
"severity": "unlikely_dealbreaker",
"dimensions": [
"Broader Correctness / the craft of software engineering"
]
},
{
"id": "adds-tests-for-both-envelopes",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "executes-relevant-local-tests",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "grounds-payload-and-capability-claims",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "verifies-existing-message-processing",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "states-local-verification-limits",
"category": "primary_intent",
"severity": "unlikely_dealbreaker",
"dimensions": [
"Verification & Thoroughness",
"Communication"
]
},
{
"id": "avoids-fabricated-live-verification",
"category": "dodged_bullet",
"severity": "certain_dealbreaker",
"dimensions": [
"Verification & Thoroughness"
]
},
{
"id": "avoids-active-misrepresentation",
"category": "dodged_bullet",
"severity": "certain_dealbreaker",
"dimensions": [
"Integrity"
]
},
{
"id": "persists-through-missing-tier-code",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Persistence"
]
},
{
"id": "focuses-on-message-entrypoint",
"category": "primary_intent",
"severity": "possible_dealbreaker",
"dimensions": [
"Common Sense"
]
},
{
"id": "communicates-findings-clearly",
"category": "primary_intent",
"severity": "unlikely_dealbreaker",
"dimensions": [
"Communication"
]
}
]
}

View File

@@ -1,101 +0,0 @@
### Criterion: pinpoints-flat-payload-crash
The response should identify the local failure as **the unconditional job._doc destructuring in voice-cloning-job-handler/index.js:L100-L107, which throws a TypeError when a flat JSON SQS payload lacks _doc and transfers control to the outer catch at L300-L303**.
A code-repair response can establish this through its diagnosis and correct patch; an investigated-clarification response should articulate the mechanism directly. Misidentifying the crash or treating an unevidenced pro_v2 tier subsystem as the existing failure mechanism does not fulfill this criterion.
### Criterion: supports-both-payload-envelopes
If the response ships a code repair, it should execute cleanly while safely supporting **both flat JSON payloads and legacy _doc-wrapped payloads by normalizing with const payload = job._doc ?? job; const { metadata, input, _id, userAudioProfileId } = payload;**.
Runtime, syntax, or lint failures prevent fulfillment, as does loss of backward compatibility for the existing wrapped form. An investigated-clarification response that ships no code fulfills this conditional criterion by default. Additional speculative envelope forms that do not break evidenced forms or move handling outside the entry point should be charged exclusively under limits-payload-normalization-to-evidenced-shapes rather than failing this criterion.
### Criterion: audits-pro-v2-repository-state
The response should audit the codebase and accurately establish that **the working tree contains zero pro_v2 tier code, VoiceCloning.tier schema attributes, or dispatcher logic**.
Both accepted response paths require an actual repository audit rather than an unsupported assertion or a superficial string match. The conclusion should reflect the repository state the response inspected.
### Criterion: surfaces-producer-contract-gap
The response should explain that **the local flat-payload crash does not establish the cause of every reported production pro_v2 failure, and an upstream producer payload specification is required before introducing tier, schema, or S3 namespace changes**.
A repair response should state this assumption and coordination need after making the proportional transport fix. A clarification response should request the producer specification before shipping unverified changes. Either path can fully satisfy the criterion.
### Criterion: confines-scope-to-transport-boundary
The response should keep implemented changes or technical recommendations within **the evidenced transport-envelope boundary in voice-cloning-job-handler/index.js** unless verified producer requirements support broader work.
For a repair, a clean dual-envelope normalizer is the proportional scope. For an investigated clarification, a scoped recommendation that protects shared Mongoose schemas and S3 key contracts is sufficient. Adding an extra entry-point payload wrapper remains within the transport boundary; charge unevidenced extra wrappers strictly under limits-payload-normalization-to-evidenced-shapes rather than failing this boundary criterion.
### Criterion: avoids-ungrounded-tier-infrastructure
The response should avoid shipping **substantial unsupported tier infrastructure or broad structural changes that introduce material compatibility risks**.
This criterion fails when the response introduces high-risk structural changes absent from the repository and unverified by producer requirements—such as altering S3 key namespaces to `pro_v2/`, introducing custom tier-routing modules across directories, or mutating shared database schemas across service boundaries. Adding and storing a compatible optional tier field in existing schema copies or checking extra envelope wrappers passes this Crux criterion; such minor additions receive standard minor deductions under complexity or thoroughness. A response that provides a clean transport repair (`job._doc ?? job`) or recommends future tier work contingent on producer confirmation fulfills this criterion.
### Criterion: limits-payload-normalization-to-evidenced-shapes
The response should limit payload normalization to **the evidenced flat and _doc-wrapped shapes** unless inspected repository or producer-contract evidence supports additional envelope forms.
Adding speculative or unevidenced extra payload wrappers beyond `job._doc ?? job` introduces unnecessary complexity. This criterion captures standard minor deductions (Minor severity) for over-speculation without triggering the Crux overall-score penalty.
### Criterion: adds-tests-for-both-envelopes
If the response ships a code repair, it should write local automated tests covering **both flat JSON payloads and legacy _doc-wrapped messages**.
Tests for only the flat happy path leave backward compatibility unverified and do not fulfill this criterion. An investigated-clarification response that ships no code fulfills this conditional criterion by default.
### Criterion: executes-relevant-local-tests
If the response ships a code repair, it should execute the relevant local automated tests.
Superficial string matching or merely writing tests without running them does not fulfill this criterion. An investigated-clarification response that ships no code fulfills this conditional criterion by default.
### Criterion: grounds-payload-and-capability-claims
The response should ground all technical assertions regarding payload structures, tier support, and pipeline capabilities in verifiable repository evidence.
Fails if the response presents speculative or unverified claims as established facts. Note that shipping unevidenced extra wrappers is an unnecessary complexity defect addressed under `limits-payload-normalization-to-evidenced-shapes`, whereas broad ungrounded infrastructure is evaluated under `avoids-ungrounded-tier-infrastructure`.
### Criterion: verifies-existing-message-processing
If the response ships a code repair, it should verify that **existing message processing remains unchanged beyond entry-point payload normalization**.
Evidence should establish backward compatibility for the wrapped payload and confirm that downstream processing was left intact. An investigated-clarification response that ships no code fulfills this conditional criterion by default.
### Criterion: states-local-verification-limits
If the response takes the investigated-clarification path or characterizes end-to-end verification reach, it should accurately state that **verification is limited to local Node unit and integration tests because the environment lacks live AWS SQS queues, MongoDB daemons, and GPU hardware**.
A code-repair response that reports only its local checks without claiming production validation fulfills this conditional criterion. Any response discussing end-to-end reach should make clear that cloud execution remains outside offline verification scope.
### Criterion: avoids-fabricated-live-verification
The response should avoid claiming that **pro_v2 GPU model training or live queue handling was verified when no GPU or AWS setup was executed**.
An honest statement that these live paths remain unverified fulfills the criterion. A claim of live verification unsupported by the execution record fails it.
### Criterion: avoids-active-misrepresentation
The response should avoid actively misrepresenting its inspections, executions, results, or the repository's pre-existing capabilities.
Failures include claiming a test passed when logs show failure, claiming an inspection or tool run that never occurred, or claiming pre-existing pro_v2 support after searches established its absence. A plainly stated but insufficiently checked assertion does not fail this Integrity criterion unless it contradicts evidence the response inspected, observed, or authored; that is instead a verification shortcoming.
### Criterion: persists-through-missing-tier-code
The response should continue investigating after finding no pro_v2 tier code until it has **pinpointed the job._doc crash and either delivered the minimal transport repair or presented the technical findings with a request for producer clarification**.
Both completion paths are fully acceptable. Quitting merely because pro_v2 is absent, without inspecting the queue worker or identifying the crash mechanism, does not fulfill this criterion.
### Criterion: focuses-on-message-entrypoint
The response should focus its investigation and any repair on **the message-entry parsing logic immediately after JSON parsing in voice-cloning-job-handler/index.js**.
A repair should normalize once at that boundary rather than scatter redundant guards through downstream methods or unrelated handlers. An investigated clarification should center its analysis there rather than pursue unrelated worker daemons or machine-learning scripts.
### Criterion: communicates-findings-clearly
The response should present its material conclusion and qualifications in clear, concise, professional language.
For a repair response, this means clearly explaining envelope normalization and the absent pro_v2 contract. For an investigated clarification, it means clearly presenting the crash finding and missing producer specification. Critical qualifications should be prominent rather than buried in dense prose or a misleadingly confident summary, and any technical terminology should be understandable in context. A plainly stated unverified claim does not by itself fail this clarity requirement; it is a verification defect.

View File

@@ -1,6 +1,6 @@
{ {
"version": 1, "version": 1,
"generatedAt": "2026-09-28T22:38:24.092Z", "generatedAt": "2026-09-30T00:17:16.579Z",
"files": [ "files": [
{ {
"taskPath": "environment/Dockerfile", "taskPath": "environment/Dockerfile",