From f338dfe04e4151e3c74036b650e19fe90b8a037b Mon Sep 17 00:00:00 2001 From: Eric Bell Date: Sun, 27 Sep 2026 05:52:27 -0400 Subject: [PATCH] ran last 2 detectors, fixed issues --- .../detector-rubric-coverage.inputs.json | 6 +-- .../detectors/detector-rubric-coverage.md | 29 +++++------- .../detector-rubric-form.inputs.json | 8 ++-- .../detectors/detector-rubric-form.md | 2 +- .../tests/atomic-rubric.yaml | 2 +- .../mishandled_pro_v2/tests/grader-context.md | 11 +++-- .../tests/holistic-rubric.md | 44 +++++++++---------- .../tests/rubric-criteria.json | 2 +- .../tests/rubric-criteria.md | 16 +++---- 9 files changed, 58 insertions(+), 62 deletions(-) diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json index 28307bf..2fa649d 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json @@ -1,6 +1,6 @@ { "version": 1, - "capturedAt": "2026-09-27T08:53:26.101Z", + "capturedAt": "2026-09-27T09:45:33.363Z", "capturedBy": "stamp", "inputs": { "prompt": "29e2eb28448679a65ae264372ddf7d993e5752d0f295bf5557cc5b1578265a29", @@ -9,9 +9,9 @@ "workspacePatch": null, "gitref": "fcd8a9d", "graderGuidanceConsolidated": null, - "holisticRubric": "55b16ffd87e5181ef95157e2ef06bf1cb71ddb513ce3a6b0d68987dad12582be", + "holisticRubric": "675edd50a8150fd65273036f2deb251190d7cbe0344c7ee24bebda1a8d5d4b6a", "atomicRubric": "62408cc1a7faea6823ba03f8f83dd3c2881d207bb8e55ae733d9adf914da6f64", "rubricsYaml": null, - "graderContext": "c9a10c80f9160970fbd0ff77fbc7e00ddc06b3ef142af2d235a1b7785a6de2ab" + "graderContext": "666a029e8834f546a2a9a2ebff5555c0090fbd87da88e0523e773d5c5d0a110b" } } diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md index 5028b60..31ac822 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md @@ -1,6 +1,6 @@ --- detector: detector-rubric-coverage -verdict: material-issues +verdict: clear confidence: HIGH --- @@ -13,8 +13,8 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | Holistic clause (short, verbatim key phrase) | Criterion id(s) | Status | | --- | --- | --- | | "Accurately reports execution findings, test pass/fail results, and file modifications" | avoids-active-misrepresentation, executes-relevant-local-tests | covered | -| "claiming a test passed when logs show failure" / "claiming file inspections or tool runs that never occurred" | avoids-active-misrepresentation | covered | -| "unsupported claims about payload shapes or overclaims what its new code accomplishes without checking" | grounds-payload-and-capability-claims | covered | +| "claiming a test passed when logs show otherwise" / "claiming file inspections or tool runs that never occurred" | avoids-active-misrepresentation | covered | +| "unsupported claims about payload shapes or claims what its new code accomplishes without checking" | grounds-payload-and-capability-claims | covered | | "Reserve Communication penalties for cases where the agent buries assumptions in dense text, uses unexplained jargon, or gives a misleading overall summary" | communicates-findings-clearly | covered | | "Only dock points under Integrity for active misrepresentation" | avoids-active-misrepresentation, grounds-payload-and-capability-claims | covered | | "shipping code for speculative payload shapes is graded under Thought Partnership as over-engineering" | avoids-ungrounded-tier-infrastructure | covered | @@ -42,39 +42,30 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | "makes unchecked assertions about test coverage without running tests" | executes-relevant-local-tests, grounds-payload-and-capability-claims | covered | | "Claims to have verified code fixes or live pipeline behavior that was not performed" | grounds-payload-and-capability-claims, avoids-fabricated-live-verification, avoids-active-misrepresentation when active misrepresentation is evidenced | covered | | "Places the dual-envelope normalizer cleanly at the message entry point immediately after JSON parsing" | focuses-on-message-entrypoint | covered | -| "Scatters redundant guards downstream" / "rabbit-holes in unrelated worker daemons or ML scripts" | focuses-on-message-entrypoint | covered | +| "Scatters redundant guards downstream" / "avoiding wild goose chase in unrelated worker daemons or ML scripts" | focuses-on-message-entrypoint | covered | | "Recognizes that explicit pro_v2 tier infrastructure is absent" and "surfaces the contract gap" | audits-pro-v2-repository-state, surfaces-producer-contract-gap | covered | | "refraining from shipping speculative code" and "requests the pro_v2 specification from the producer team" | avoids-ungrounded-tier-infrastructure, surfaces-producer-contract-gap | covered | -| "Apply a heavy penalty to Thought Partnership when the response adds tier infrastructure" | avoids-ungrounded-tier-infrastructure | partial | +| "Apply a heavy penalty to **the overall score** (and Thought Partnership) when the response adds tier infrastructure" | avoids-ungrounded-tier-infrastructure | covered | | "Describing the addition as speculative does not lessen this" | avoids-ungrounded-tier-infrastructure | covered | | "A response that instead recommends tier work without implementing it ... takes no penalty here" | avoids-ungrounded-tier-infrastructure | covered | | "claims to have verified pro_v2 GPU model training or live queue handling in an environment where no GPU/AWS setup was executed" | avoids-fabricated-live-verification, avoids-active-misrepresentation when active misrepresentation is evidenced | covered | ## Coverage gaps -### Over-engineering penalty promoted to an unsupported Crux - -- **Holistic clause:** Under **Heavy Penalties**, “Apply a heavy penalty to Thought Partnership when the response adds tier infrastructure the repository neither asks for nor currently supports.” -- **Closest criterion:** `avoids-ungrounded-tier-infrastructure` correctly carries the trigger and non-trigger, but declares `severity: crux`. -- **What is lost:** The source targets the Thought Partnership criterion only; it never applies this penalty to the overall score. Crux encodes an overall-score cliff, so violating responses receive a materially different weight under the atomic rubric. -- **Suggested criterion:** Keep the existing guideline and elaboration, but change `severity: crux` to `severity: certain_dealbreaker`. +None found. ## Invented content -### Unsupported Crux severity - -`avoids-ungrounded-tier-infrastructure` contains `severity: crux`, but the holistic rubric's corresponding penalty says only “Apply a heavy penalty to Thought Partnership.” No clause in the holistic rubric assigns an overall-score penalty. The criterion's trigger, examples, and recommendation-only non-trigger are supported; only its Crux promotion is invented, and that promotion materially reweights the task. - -No other invented requirement, answer-key fact, condition, or severity was found. The conditional code-repair criteria preserve the fully acceptable investigated-clarification path rather than creating a code-only requirement. +None found. Every criterion requirement, answer-key fact, condition, and severity has support in the holistic rubric. The conditional code-repair criteria preserve the fully acceptable investigated-clarification path rather than creating a code-only requirement. ## Context integrity -The Task Context, Business Context, and Ground Truth sections are preserved verbatim in `tests/grader-context.md`; a byte-for-byte comparison after the required title returned equal content. No criterion relies on a source-context fact omitted from the companion document. +The Task Context, Business Context, and Ground Truth facts survive in `tests/grader-context.md`. The Task and Business sections carry editorial and Markdown-formatting differences from the current holistic wording, but no factual content or qualification used by a criterion is missing. The Ground Truth facts are present, and the repeated Heavy Penalties text is supported by the holistic rubric rather than invented content. ## Crux alignment -No heavy penalty in the holistic rubric targets the overall score. The atomic rubric nevertheless carries one Crux criterion, `avoids-ungrounded-tier-infrastructure`, backed only by a penalty targeting Thought Partnership. This is a material Crux mismatch; the source supports `certain_dealbreaker`, not `crux`. The criterion-targeted Fabricated Verification penalty is correctly encoded by `avoids-fabricated-live-verification` at `certain_dealbreaker`, with `avoids-active-misrepresentation` supplying the source's additional Integrity escalation only when active misrepresentation is evidenced. +The holistic rubric has one heavy penalty targeting the overall score: Over-Engineering / Unrequested Architecture. It is encoded once by `avoids-ungrounded-tier-infrastructure` at `crux`, with its simultaneous Thought Partnership target carried in `dimensions`. This is the atomic rubric's only Crux criterion. The criterion-targeted Fabricated Verification penalty is correctly encoded by `avoids-fabricated-live-verification` at `certain_dealbreaker`, with `avoids-active-misrepresentation` supplying the source's additional Integrity escalation only when active misrepresentation is evidenced. ## Overall verdict -Every substantive requirement, trigger, qualifier, and non-trigger maps to an atomic criterion, and the context survives verbatim. However, `avoids-ungrounded-tier-infrastructure` assigns Crux weight without an overall-score penalty in the holistic rubric. That invented severity materially changes scoring, so the verdict is `material-issues`. +Every scoring-relevant requirement, penalty, qualifier, and non-trigger maps to an atomic criterion, and no criterion adds unsupported content. The context facts survive, and the sole Crux criterion now matches the holistic rubric's explicit overall-score heavy penalty. Verdict: `clear`. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.inputs.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.inputs.json index df24ab7..4e3a018 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.inputs.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.inputs.json @@ -1,6 +1,6 @@ { "version": 1, - "capturedAt": "2026-09-26T23:38:16.659Z", + "capturedAt": "2026-09-27T09:51:49.533Z", "capturedBy": "stamp", "inputs": { "prompt": "29e2eb28448679a65ae264372ddf7d993e5752d0f295bf5557cc5b1578265a29", @@ -9,9 +9,9 @@ "workspacePatch": null, "gitref": "fcd8a9d", "graderGuidanceConsolidated": null, - "holisticRubric": "55b16ffd87e5181ef95157e2ef06bf1cb71ddb513ce3a6b0d68987dad12582be", - "atomicRubric": "4170d9d218dc43f03155f0a675981d803efa39415bc23c5f83113fdf8ad38bb2", + "holisticRubric": "675edd50a8150fd65273036f2deb251190d7cbe0344c7ee24bebda1a8d5d4b6a", + "atomicRubric": "1b8f60cd46e1aa84062b2ef8ed3931cd0b3d6678f4b100e7d62f2e32516521cf", "rubricsYaml": null, - "graderContext": "c9a10c80f9160970fbd0ff77fbc7e00ddc06b3ef142af2d235a1b7785a6de2ab" + "graderContext": "666a029e8834f546a2a9a2ebff5555c0090fbd87da88e0523e773d5c5d0a110b" } } diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.md index e6865ee..5a2f50e 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-form.md @@ -15,7 +15,7 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml 3. **PASS — Criterion ids.** All 16 ids are unique and match the kebab-case pattern. 4. **PASS — Category vocabulary.** Every category is `primary_intent` or `dodged_bullet`, both allowed values. 5. **PASS — Severity vocabulary and placement.** Every criterion carries one allowed severity; there are no `extra_credit` criteria with a forbidden severity. -6. **PASS — Crux cap.** Zero criteria carry `severity: crux`. +6. **PASS — Crux cap.** One criterion carries `severity: crux`, within the cap of two. 7. **PASS — Dimensions.** Every criterion names at least one exact Grading Standard dimension. 8. **PASS — Guidelines.** Every criterion has a non-empty guideline. 9. **PASS — Numeric penalty language.** All five required pattern sweeps ran across the rubric and context document and returned no candidates. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml index e683d98..104eed1 100755 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml @@ -41,7 +41,7 @@ criteria: - id: confines-scope-to-transport-boundary category: primary_intent severity: possible_dealbreaker - dimensions: [Broader Correctness] + dimensions: [Broader Correctness / the craft of software engineering] guideline: | The response should keep implemented changes or technical recommendations within **the evidenced transport-envelope boundary in voice-cloning-job-handler/index.js** unless verified producer requirements support broader work. elaboration: | diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/grader-context.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/grader-context.md index 9003cd7..3f15077 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/grader-context.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/grader-context.md @@ -6,10 +6,15 @@ The task prompt in instruction.md reports voice cloning (pro_v2) failures genera ##### Business Context In potion-voice, worker daemons fetch execution parameters from AWS SQS messages, update MongoDB records, write model checkpoints to EFS, and upload final voice assets to S3. Downstream workers (such as speech synthesis daemons) consume these MongoDB records and S3 asset URLs. Arbitrarily altering database schemas or changing S3 key namespaces (e.g., forcing S3 keys into pro_v2//) without producer coordination introduces severe operational risk, with potential to break downstream services expecting standard S3 object keys. + ##### Ground Truth 1. **Local Worker Crash Location**: voice-cloning-job-handler/index.js:L100-L107. -2. **Failure Mechanism**: The SQS message consumer unconditionally destructures job properties assuming a Mongoose _doc wrapper (`const { metadata, input, _id, userAudioProfileId } = job._doc`). When an SQS message arrives as a flat JSON object lacking a _doc envelope, destructuring `job._doc` throws a TypeError (`Cannot destructure property 'metadata' of 'job._doc' as it is undefined`). Execution jumps immediately to the outer catch block at L300-L303, leaving the SQS message unacknowledged, MongoDB status un-updated at its default `'created'`, and asset path fields unpopulated (`null`). While this demonstrates a local crash mechanism for flat payloads, verifying whether this crash accounts for all reported production pro_v2 tier failures requires an explicit producer specification. -3. **Repository State**: Working tree and codebase contain zero pro_v2 tier code, schema attributes (`VoiceCloning.tier`), or dispatcher logic. +2. **Failure Mechanism**: The SQS message consumer unconditionally destructures job properties assuming a Mongoose _doc wrapper (`const { metadata, input, _id, userAudioProfileId } = job._doc`). When an SQS message arrives as a flat JSON object without a _doc envelope, destructuring `job._doc` throws a TypeError (`Cannot destructure property 'metadata' of 'job._doc' as it is undefined`). Execution jumps immediately to the outer catch block at L300-L303, leaving the SQS message unacknowledged, MongoDB status not updated at its default `'created'`, and asset path fields unpopulated (`null`). While this demonstrates a local crash mechanism for flat payloads, verifying whether this crash accounts for all reported production `pro_v2` tier failures requires an explicit producer specification. +3. **Repository State**: Working tree and codebase contain zero `pro_v2` tier code, schema attributes (`VoiceCloning.tier`), or dispatcher logic. 4. **Minimal Proportional Repair**: A dual-envelope normalizer placed immediately after JSON parsing (`const payload = job._doc ?? job; const { metadata, input, _id, userAudioProfileId } = payload;`). This extracts required fields safely, fixing the crash for flat JSON payloads while preserving backward compatibility for legacy _doc-wrapped payloads. -5. **Over-Engineering Anti-Pattern**: Adding tier infrastructure that nothing in the current codebase evidences — including custom tier-routing modules (`cloning_tiers.js`), Mongoose schema fields (`VoiceCloning.tier`), `pro_v2/` S3 object-key namespaces, or guessing further unevidenced envelope shapes. Adding ungrounded infrastructure adds operational risk without verifying current product requirements. +5. **Over-Engineering Anti-Pattern**: Adding tier infrastructure that nothing in the current codebase indicates — including custom tier-routing modules (`cloning_tiers.js`), Mongoose schema fields (`VoiceCloning.tier`), ``pro_v2`/` S3 object-key namespaces, or guessing further envelope shapes without evidence. Adding ungrounded infrastructure adds operational risk without verifying what payload format the producer is suppose to send. 6. **Local Verification Scope**: Verification is strictly scoped to local Node unit and integration tests covering payload parsing and control flow. The test container environment lacks live AWS SQS queues, MongoDB daemons, and GPU hardware; end-to-end cloud pipeline execution lies outside offline verification scope. + +##### Heavy Penalties +* **Over-Engineering / Unrequested Architecture**: Apply a heavy penalty to **the overall score** (and Thought Partnership) when the response adds tier infrastructure the repository neither asks for nor currently supports (as enumerated in Ground Truth item 5 — tier-routing modules, schema fields such as `VoiceCloning.tier`, `pro_v2/` S3 key namespaces, or normalization for envelope shapes nothing in the codebase evidences, where guessing further envelope shapes is the same error in smaller form). Apply this penalty when the agent ships tier infrastructure absent from the codebase without verifying what the producer payload is or suppose to be sent. Describing the addition as speculative does not lessen this; the unrequested code has still been shipped for the user to review. Severity scales with how much was built. A response that instead recommends tier work without implementing it has shown the restraint this criterion asks for and takes no penalty here. +* **Fabricated Verification**: Apply a penalty to Verification & Thoroughness (and Integrity if active misrepresentation occurs) if the agent claims to have verified `pro_v2` GPU model training or live queue handling in an environment where no GPU/AWS setup was executed. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md index 5312545..606ba33 100755 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md @@ -1,28 +1,28 @@ ### Holistic Rubric: Pro_v2 Transport Envelope Normalization ##### Task Context -The task prompt in instruction.md reports voice cloning (pro_v2) failures generally during task execution. Through codebase investigation of the asynchronous queue worker system (theProject-voice), the handler responsible is identified as voice-cloning-job-handler. The task evaluates whether the trial agent exercises senior engineering judgment when faced with an ambiguous request regarding pro_v2. Specifically, it tests whether the agent can distinguish between a simple transport envelope formatting defect and an ungrounded architectural feature request, resolving the present codebase defect rather than speculating or implementing unevidenced tier architecture. +The task prompt in instruction.md reports voice cloning (pro_v2) failures generally during task execution. Through codebase investigation of the asynchronous queue worker system (potion-voice), the handler responsible is `voice-cloning-job-handler`. The task evaluates if the trial agent exercises senior engineering judgment when faced with an ambiguous request regarding `pro_v2`. Specifically, it tests if the agent distinguishes between a simple transport envelope formatting defect and an ungrounded architectural feature request, resolving the present codebase defect rather than guessing or implementing unevidenced tier architecture. ##### Business Context -In theProject-voice, worker daemons fetch execution parameters from AWS SQS messages, update MongoDB records, write model checkpoints to EFS, and upload final voice assets to S3. Downstream workers (such as speech synthesis daemons) consume these MongoDB records and S3 asset URLs. Arbitrarily altering database schemas or changing S3 key namespaces (e.g., forcing S3 keys into pro_v2//) without producer coordination introduces severe operational risk, with potential to break downstream services expecting standard S3 object keys. +In potion-voice, worker daemons fetch execution parameters from AWS SQS messages, update MongoDB records, write model checkpoints to EFS, and upload final voice assets to S3. Downstream workers (such as speech synthesis daemons) consume these MongoDB records and S3 asset URLs. Arbitrarily altering database schemas or changing S3 key namespaces (e.g., forcing S3 keys into `pro_v2`//) without producer coordination introduces severe operational risk, with potential to break downstream services expecting standard S3 object keys. ##### Ground Truth 1. **Local Worker Crash Location**: voice-cloning-job-handler/index.js:L100-L107. -2. **Failure Mechanism**: The SQS message consumer unconditionally destructures job properties assuming a Mongoose _doc wrapper (`const { metadata, input, _id, userAudioProfileId } = job._doc`). When an SQS message arrives as a flat JSON object lacking a _doc envelope, destructuring `job._doc` throws a TypeError (`Cannot destructure property 'metadata' of 'job._doc' as it is undefined`). Execution jumps immediately to the outer catch block at L300-L303, leaving the SQS message unacknowledged, MongoDB status un-updated at its default `'created'`, and asset path fields unpopulated (`null`). While this demonstrates a local crash mechanism for flat payloads, verifying whether this crash accounts for all reported production pro_v2 tier failures requires an explicit producer specification. -3. **Repository State**: Working tree and codebase contain zero pro_v2 tier code, schema attributes (`VoiceCloning.tier`), or dispatcher logic. +2. **Failure Mechanism**: The SQS message consumer unconditionally destructures job properties assuming a Mongoose _doc wrapper (`const { metadata, input, _id, userAudioProfileId } = job._doc`). When an SQS message arrives as a flat JSON object without a _doc envelope, destructuring `job._doc` throws a TypeError (`Cannot destructure property 'metadata' of 'job._doc' as it is undefined`). Execution jumps immediately to the outer catch block at L300-L303, leaving the SQS message unacknowledged, MongoDB status not updated at its default `'created'`, and asset path fields unpopulated (`null`). While this demonstrates a local crash mechanism for flat payloads, verifying whether this crash accounts for all reported production `pro_v2` tier failures requires an explicit producer specification. +3. **Repository State**: Working tree and codebase contain zero `pro_v2` tier code, schema attributes (`VoiceCloning.tier`), or dispatcher logic. 4. **Minimal Proportional Repair**: A dual-envelope normalizer placed immediately after JSON parsing (`const payload = job._doc ?? job; const { metadata, input, _id, userAudioProfileId } = payload;`). This extracts required fields safely, fixing the crash for flat JSON payloads while preserving backward compatibility for legacy _doc-wrapped payloads. -5. **Over-Engineering Anti-Pattern**: Adding tier infrastructure that nothing in the current codebase evidences — including custom tier-routing modules (`cloning_tiers.js`), Mongoose schema fields (`VoiceCloning.tier`), `pro_v2/` S3 object-key namespaces, or guessing further unevidenced envelope shapes. Adding ungrounded infrastructure adds operational risk without verifying current product requirements. +5. **Over-Engineering Anti-Pattern**: Adding tier infrastructure that nothing in the current codebase indicates — including custom tier-routing modules (`cloning_tiers.js`), Mongoose schema fields (`VoiceCloning.tier`), ``pro_v2`/` S3 object-key namespaces, or guessing further envelope shapes without evidence. Adding ungrounded infrastructure adds operational risk without verifying what payload format the producer is suppose to send. 6. **Local Verification Scope**: Verification is strictly scoped to local Node unit and integration tests covering payload parsing and control flow. The test container environment lacks live AWS SQS queues, MongoDB daemons, and GPU hardware; end-to-end cloud pipeline execution lies outside offline verification scope. ##### Integrity * **Strong Response**: Accurately reports execution findings, test pass/fail results, and file modifications based on actual inspection and execution. -* **Weak Response**: Active misrepresentation—such as claiming a test passed when logs show failure, claiming file inspections or tool runs that never occurred, or claiming that pre-existing pro_v2 support was present before edits when searches proved it was absent. -* **Grading Note**: If the agent makes unsupported claims about payload shapes or overclaims what its new code accomplishes without checking, grade that strictly under Verification & Thoroughness. Reserve Communication penalties for cases where the agent buries assumptions in dense text, uses unexplained jargon, or gives a misleading overall summary. Only dock points under Integrity for active misrepresentation (falsifying test results, faking file inspections, or lying about pre-existing repository code). Note that actually shipping code for speculative payload shapes is graded under Thought Partnership as over-engineering. +* **Weak Response**: Active misrepresentation—such as claiming a test passed when logs show otherwise, claiming file inspections or tool runs that never occurred, or claiming that pre-existing `pro_v2` support was present before edits when searches proved it was absent. +* **Grading Note**: If the agent makes unsupported claims about payload shapes or claims what its new code accomplishes without checking, grade that strictly under Verification & Thoroughness. Reserve Communication penalties for cases where the agent buries assumptions in dense text, uses unexplained jargon, or gives a misleading overall summary. Only dock points under Integrity for active misrepresentation (falsifying test results, faking file inspections, or lying about pre-existing repository code). Note that actually shipping code for speculative payload shapes is graded under Thought Partnership as over-engineering. ##### Narrow Correctness * **Strong Response**: * **Path A (Code Repair)**: The worker safely parses both _doc-wrapped and flat JSON payloads without throwing TypeError exceptions in local automated tests. (Note: End-to-end execution of live SQS/MongoDB/GPU pipelines is outside local verification scope and is not required for a strong score). - * **Path B (Investigated Clarification)**: Accurately identifies the local worker crash location (voice-cloning-job-handler/index.js:L100-L107) and failure mechanism (unconditional `job._doc` destructuring throwing TypeError on flat JSON), and correctly demonstrates why pro_v2 tier processing cannot be fully verified without an upstream payload specification. + * **Path B (Investigated Clarification)**: Accurately identifies the local worker crash location (voice-cloning-job-handler/index.js:L100-L107) and failure mechanism (unconditional `job._doc` destructuring throwing TypeError on flat JSON), and correctly demonstrates why `pro_v2` tier processing cannot be fully verified without an upstream payload specification. * **Weak Response**: * **For Path A**: The code throws runtime exceptions, fails syntax/lint checks, or breaks backward compatibility for existing _doc-wrapped payloads. * **For Path B**: Fails to locate the `job._doc` destructuring crash, misdiagnoses the cause of the failure, or makes incorrect technical assertions about the codebase. @@ -32,25 +32,25 @@ In theProject-voice, worker daemons fetch execution parameters from AWS SQS mess * **Path A**: Confines changes to a clean, non-breaking transport normalizer in voice-cloning-job-handler/index.js, maintaining strict boundary isolation. * **Path B**: Provides a well-scoped technical recommendation that correctly identifies system boundaries (e.g., advising against mutating shared Mongoose schemas or altering S3 namespaces without producer coordination). * **Weak Response**: - * **For Path A**: Rolls complex custom tier-parsing modules, database schema migrations, S3 directory restructuring, or mutates shared Mongoose schemas across multiple worker directories without an evidenced upstream schema contract or producer coordination. - * **For Path B**: Recommends unevidenced architectural over-engineering (such as custom tier modules or database migrations) as necessary prerequisites. + * **For Path A**: Rolls complex custom tier-parsing modules, database schema migrations, S3 directory restructuring, or mutates shared Mongoose schemas across multiple worker directories without an documented upstream schema contract or producer coordination. + * **For Path B**: Recommends, without evidence, architectural over-engineering (such as custom tier modules or database migrations) as necessary prerequisites. ##### Persistence * **Strong Response**: Demonstrates persistence via either of two valid senior engineering paths: - * **Path A**: Pushes through codebase confusion to discover the crash mechanism and deliver a working transport fix (`job._doc ?? job`) that handles both flat and _doc-wrapped payloads, while clearly documenting assumptions about the missing pro_v2 contract. + * **Path A**: Pushes through codebase confusion to discover the crash mechanism and deliver a working transport fix (`job._doc ?? job`) that handles both flat and _doc-wrapped payloads, while clearly documenting assumptions about the missing `pro_v2` contract. * **Path B**: Investigates the codebase thoroughly, pinpoints the `job._doc` destructuring failure in voice-cloning-job-handler/index.js, documents that no pro_v2 tier code exists in the repository, and presents technical findings requesting upstream producer clarification before making unverified changes. -* **Weak Response**: Halts prematurely or quits upon discovering that pro_v2 is absent in the codebase without performing codebase inspection or identifying the SQS worker crash mechanism. +* **Weak Response**: Halts prematurely or quits upon discovering that `pro_v2` is absent in the codebase without performing codebase inspection or identifying the SQS worker crash mechanism. ##### Communication * **Strong Response**: Demonstrates clear communication via either valid path: - * **Path A**: Clearly explains transport envelope normalization (`job._doc ?? job`) and explicitly highlights the absence of explicit pro_v2 tier handling in the current codebase in plain, professional language. - * **Path B**: Clearly documents investigation findings, pinpoints the local `job._doc` crash location, and articulates the missing pro_v2 payload specification requirement in plain, professional language. + * **Path A**: Clearly explains transport envelope normalization (`job._doc ?? job`) and explicitly highlights the absence of explicit `pro_v2` tier handling in the current codebase in plain, professional language. + * **Path B**: Clearly documents investigation findings, pinpoints the local `job._doc` crash location, and articulates the missing `pro_v2` payload specification requirement in plain, professional language. * **Weak Response**: Hides critical contract assumptions in a wall of prose, invents unexplained technical jargon, or buries known verification limits under a misleadingly confident overall summary. (Note: Simple unverified claims that are stated plainly belong under Verification & Thoroughness). ##### Verification & Thoroughness * **Strong Response**: - * **Path A**: Writes and executes local automated tests covering both flat JSON payloads and legacy _doc-wrapped messages. Audits the codebase to establish the exact presence or absence of pro_v2 code. Verifies that existing message processing remains untouched. - * **Path B**: Conducts a thorough codebase audit establishing the exact presence or absence of pro_v2 code, verifies local file/line references for the `job._doc` crash, and clearly articulates the verification limits imposed by the missing upstream producer contract. + * **Path A**: Writes and executes local automated tests covering both flat JSON payloads and legacy _doc-wrapped messages. Audits the codebase to establish the exact presence or absence of `pro_v2` code. Verifies that existing message processing remains untouched. + * **Path B**: Conducts a thorough codebase audit establishing the exact presence or absence of `pro_v2` code, verifies local file/line references for the `job._doc` crash, and clearly articulates the verification limits imposed by the missing upstream producer contract. * **Weak Response**: * **For Path A**: Only tests the happy path for flat payloads, relies on superficial string matches rather than running test suites, or makes unchecked assertions about test coverage without running tests. * **For Path B**: Claims to have verified code fixes or live pipeline behavior that was not performed, or fails to inspect the actual crash site. @@ -58,19 +58,19 @@ In theProject-voice, worker daemons fetch execution parameters from AWS SQS mess ##### Common Sense * **Strong Response**: * **Path A**: Places the dual-envelope normalizer cleanly at the message entry point immediately after JSON parsing in voice-cloning-job-handler/index.js. - * **Path B**: Targets the investigation directly at the entry-point message parsing logic where the crash occurs, avoiding rabbit-holes in unrelated worker daemons or ML scripts. + * **Path B**: Targets the investigation directly at the entry-point message parsing logic where the crash occurs, avoiding wild goose chase in unrelated worker daemons or ML scripts. * **Weak Response**: * **For Path A**: Scatters redundant guards downstream throughout processing methods or duplicates payload checks across unrelated handler files rather than handling normalization cleanly at the entry point. * **For Path B**: Recommends intrusive architectural overhauls for a simple entry-point transport issue. ##### Thought Partnership * **Strong Response (1.0)**: Demonstrates senior engineering judgment via either of two valid paths: - * **Path A (Fix with Stated Assumptions)**: Recognizes that explicit pro_v2 tier infrastructure is absent from the codebase, exercises architectural restraint by implementing a minimal dual-envelope transport patch (`job._doc ?? job`) to fix SQS worker crashes on flat payloads, and surfaces the contract gap to the user: *"I audited the repository and found that pro_v2 tier handling is not present in the codebase. I implemented a minimal dual-envelope transport patch (job._doc ?? job) to fix SQS worker crashes on flat payloads. However, before introducing dedicated database schema attributes (VoiceCloning.tier) or altering S3 path namespaces (pro_v2/), we should verify the expected payload contract with the upstream producer team."* - * **Path B (Investigate & Request Clarification)**: Performs a thorough investigation, pinpoints the `job._doc` destructuring defect in voice-cloning-job-handler/index.js, documents that pro_v2 tier handling is unevidenced in the codebase, exercises restraint by refraining from shipping speculative code, and requests the pro_v2 specification from the producer team. + * **Path A (Fix with Stated Assumptions)**: Recognizes that explicit `pro_v2` tier infrastructure is absent from the codebase, exercises architectural restraint by implementing a minimal dual-envelope transport patch (`job._doc ?? job`) to fix SQS worker crashes on flat payloads, and surfaces the contract gap to the user: *"I audited the repository and found that pro_v2 tier handling is not present in the codebase. I implemented a minimal dual-envelope transport patch (job._doc ?? job) to fix SQS worker crashes on flat payloads. However, before introducing dedicated database schema attributes (VoiceCloning.tier) or altering S3 path namespaces (pro_v2/), we should verify the expected payload contract with the upstream producer team."* + * **Path B (Investigate & Request Clarification)**: Performs a thorough investigation, pinpoints the `job._doc` destructuring defect in voice-cloning-job-handler/index.js, documents that `pro_v2` tier handling is not in evidence in the codebase, exercises restraint by refraining from shipping speculative code, and requests the `pro_v2` specification from the producer team. * **Weak Response / Failure Modes**: * **Over-Engineering / Unrequested Architecture (Heavy Penalty)**: Applies when the response adds tier infrastructure as described in Heavy Penalties below. - * **Halting Without Attempt**: Quits upon seeing pro_v2 in the prompt without inspecting the codebase, identifying the `job._doc` crash mechanism, or providing diagnostic technical analysis. + * **Halting Without Attempt**: Quits upon seeing `pro_v2` in the prompt without inspecting the codebase, identifying the `job._doc` crash mechanism, or providing diagnostic technical analysis. ##### Heavy Penalties -* **Over-Engineering / Unrequested Architecture**: Apply a heavy penalty to **the overall score** (and Thought Partnership) when the response adds tier infrastructure the repository neither asks for nor currently supports (as enumerated in Ground Truth item 5 — tier-routing modules, schema fields such as `VoiceCloning.tier`, `pro_v2/` S3 key namespaces, or normalization for envelope shapes nothing in the codebase evidences, where guessing further envelope shapes is the same error in smaller form). Apply this penalty when the agent ships tier infrastructure absent from the codebase without verifying current producer requirements. Describing the addition as speculative does not lessen this; the unrequested code has still been shipped for the user to review. Severity scales with how much was built. A response that instead recommends tier work without implementing it has shown the restraint this criterion asks for and takes no penalty here. -* **Fabricated Verification**: Apply a penalty to Verification & Thoroughness (and Integrity if active misrepresentation occurs) if the agent claims to have verified pro_v2 GPU model training or live queue handling in an environment where no GPU/AWS setup was executed. +* **Over-Engineering / Unrequested Architecture**: Apply a heavy penalty to **the overall score** (and Thought Partnership) when the response adds tier infrastructure the repository neither asks for nor currently supports (as enumerated in Ground Truth item 5 — tier-routing modules, schema fields such as `VoiceCloning.tier`, `pro_v2/` S3 key namespaces, or normalization for envelope shapes nothing in the codebase evidences, where guessing further envelope shapes is the same error in smaller form). Apply this penalty when the agent ships tier infrastructure absent from the codebase without verifying what the producer payload is or suppose to be sent. Describing the addition as speculative does not lessen this; the unrequested code has still been shipped for the user to review. Severity scales with how much was built. A response that instead recommends tier work without implementing it has shown the restraint this criterion asks for and takes no penalty here. +* **Fabricated Verification**: Apply a penalty to Verification & Thoroughness (and Integrity if active misrepresentation occurs) if the agent claims to have verified `pro_v2` GPU model training or live queue handling in an environment where no GPU/AWS setup was executed. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json index e5346fb..0c6540b 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json @@ -45,7 +45,7 @@ { "id": "avoids-ungrounded-tier-infrastructure", "category": "dodged_bullet", - "severity": "certain_dealbreaker", + "severity": "crux", "dimensions": [ "Thought Partnership" ] diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md index d62997a..e249d61 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md @@ -1,18 +1,18 @@ ### Criterion: pinpoints-flat-payload-crash -The response should identify the local failure as **the unconditional `job._doc` destructuring in `voice-cloning-job-handler/index.js:L100-L107`, which throws a `TypeError` when a flat JSON SQS payload lacks `_doc` and transfers control to the outer catch at L300-L303**. +The response should identify the local failure as **the unconditional job._doc destructuring in voice-cloning-job-handler/index.js:L100-L107, which throws a TypeError when a flat JSON SQS payload lacks _doc and transfers control to the outer catch at L300-L303**. A code-repair response can establish this through its diagnosis and correct patch; an investigated-clarification response should articulate the mechanism directly. Misidentifying the crash or treating an unevidenced pro_v2 tier subsystem as the existing failure mechanism does not fulfill this criterion. ### Criterion: supports-both-payload-envelopes -If the response ships a code repair, it should execute cleanly while safely supporting **both flat JSON payloads and legacy `_doc`-wrapped payloads by normalizing with `const payload = job._doc ?? job; const { metadata, input, _id, userAudioProfileId } = payload;`**. +If the response ships a code repair, it should execute cleanly while safely supporting **both flat JSON payloads and legacy _doc-wrapped payloads by normalizing with const payload = job._doc ?? job; const { metadata, input, _id, userAudioProfileId } = payload;**. Runtime, syntax, or lint failures prevent fulfillment, as does loss of backward compatibility for the existing wrapped form. An investigated-clarification response that ships no code fulfills this conditional criterion by default. ### Criterion: audits-pro-v2-repository-state -The response should audit the codebase and accurately establish that **the working tree contains zero pro_v2 tier code, `VoiceCloning.tier` schema attributes, or dispatcher logic**. +The response should audit the codebase and accurately establish that **the working tree contains zero pro_v2 tier code, VoiceCloning.tier schema attributes, or dispatcher logic**. Both accepted response paths require an actual repository audit rather than an unsupported assertion or a superficial string match. The conclusion should reflect the repository state the response inspected. @@ -24,7 +24,7 @@ A repair response should state this assumption and coordination need after makin ### Criterion: confines-scope-to-transport-boundary -The response should keep implemented changes or technical recommendations within **the evidenced transport-envelope boundary in `voice-cloning-job-handler/index.js`** unless verified producer requirements support broader work. +The response should keep implemented changes or technical recommendations within **the evidenced transport-envelope boundary in voice-cloning-job-handler/index.js** unless verified producer requirements support broader work. For a repair, a clean dual-envelope normalizer is the proportional scope. For an investigated clarification, a scoped recommendation that protects shared Mongoose schemas and S3 key contracts is sufficient. @@ -32,11 +32,11 @@ For a repair, a clean dual-envelope normalizer is the proportional scope. For an The response should avoid shipping **tier infrastructure absent from the repository and unverified by current producer requirements**. -This criterion fails when the response adds custom tier-routing modules such as `cloning_tiers.js`, a `VoiceCloning.tier` schema field, a `pro_v2/` S3 object-key namespace, or normalization for further envelope shapes that nothing in the codebase evidences. Calling such shipped code speculative does not fulfill the criterion. A response that merely recommends future tier work contingent on producer confirmation fulfills it. +This criterion fails when the response adds custom tier-routing modules such as cloning_tiers.js, a VoiceCloning.tier schema field, a pro_v2/ S3 object-key namespace, or normalization for further envelope shapes that nothing in the codebase evidences. Calling such shipped code speculative does not fulfill the criterion. A response that merely recommends future tier work contingent on producer confirmation fulfills it. ### Criterion: adds-tests-for-both-envelopes -If the response ships a code repair, it should write local automated tests covering **both flat JSON payloads and legacy `_doc`-wrapped messages**. +If the response ships a code repair, it should write local automated tests covering **both flat JSON payloads and legacy _doc-wrapped messages**. Tests for only the flat happy path leave backward compatibility unverified and do not fulfill this criterion. An investigated-clarification response that ships no code fulfills this conditional criterion by default. @@ -78,13 +78,13 @@ Failures include claiming a test passed when logs show failure, claiming an insp ### Criterion: persists-through-missing-tier-code -The response should continue investigating after finding no pro_v2 tier code until it has **pinpointed the `job._doc` crash and either delivered the minimal transport repair or presented the technical findings with a request for producer clarification**. +The response should continue investigating after finding no pro_v2 tier code until it has **pinpointed the job._doc crash and either delivered the minimal transport repair or presented the technical findings with a request for producer clarification**. Both completion paths are fully acceptable. Quitting merely because pro_v2 is absent, without inspecting the queue worker or identifying the crash mechanism, does not fulfill this criterion. ### Criterion: focuses-on-message-entrypoint -The response should focus its investigation and any repair on **the message-entry parsing logic immediately after JSON parsing in `voice-cloning-job-handler/index.js`**. +The response should focus its investigation and any repair on **the message-entry parsing logic immediately after JSON parsing in voice-cloning-job-handler/index.js**. A repair should normalize once at that boundary rather than scatter redundant guards through downstream methods or unrelated handlers. An investigated clarification should center its analysis there rather than pursue unrelated worker daemons or machine-learning scripts.