From a1c9c5fe95f528e0d689d36f77c1a33d6ea08e5d Mon Sep 17 00:00:00 2001 From: Eric Bell Date: Mon, 28 Sep 2026 14:13:45 -0400 Subject: [PATCH] ran coverage, fixed issue, got clear value --- .../detector-rubric-coverage.inputs.json | 8 +++---- .../detectors/detector-rubric-coverage.md | 24 +++++++++++-------- .../tests/atomic-rubric.yaml | 9 +++++++ .../tests/rubric-criteria.json | 8 +++++++ .../tests/rubric-criteria.md | 6 +++++ 5 files changed, 41 insertions(+), 14 deletions(-) diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json index de1dbb8..64030b5 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json @@ -1,6 +1,6 @@ { "version": 1, - "capturedAt": "2026-09-27T11:42:39.035Z", + "capturedAt": "2026-09-28T18:12:01.032Z", "capturedBy": "stamp", "inputs": { "prompt": "29e2eb28448679a65ae264372ddf7d993e5752d0f295bf5557cc5b1578265a29", @@ -9,9 +9,9 @@ "workspacePatch": null, "gitref": "fcd8a9d", "graderGuidanceConsolidated": null, - "holisticRubric": "675edd50a8150fd65273036f2deb251190d7cbe0344c7ee24bebda1a8d5d4b6a", - "atomicRubric": "1b8f60cd46e1aa84062b2ef8ed3931cd0b3d6678f4b100e7d62f2e32516521cf", + "holisticRubric": "d1842656433024e283f2732909cbc9a58f924adefe80f5c4ebef7544551f7522", + "atomicRubric": "3d3dc4ffe395c1fd3887d275145d8846345c4d93fdbc15cc5a49ee2419b24ed1", "rubricsYaml": null, - "graderContext": "666a029e8834f546a2a9a2ebff5555c0090fbd87da88e0523e773d5c5d0a110b" + "graderContext": "c0924eff78105ed8e16a8f8fcf81594e51146f39d7a5788b462768b1d0ce6653" } } diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md index 53a5d60..467b9e2 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md @@ -17,7 +17,7 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | "unsupported claims about payload shapes or claims what its new code accomplishes without checking" | grounds-payload-and-capability-claims | covered | | "Reserve Communication penalties for cases where the agent buries assumptions in dense text, uses unexplained jargon, or gives a misleading overall summary" | communicates-findings-clearly | covered | | "Only dock points under Integrity for active misrepresentation" | avoids-active-misrepresentation, grounds-payload-and-capability-claims | covered | -| "shipping code for speculative payload shapes is graded under Thought Partnership as over-engineering" | avoids-ungrounded-tier-infrastructure | covered | +| "actually shipping code for speculative payload shapes is graded under Thought Partnership as over-engineering" and minor "extra payload wrapper" additions "should receive standard deductions under Thought Partnership or Broader Correctness" | limits-payload-normalization-to-evidenced-shapes, avoids-ungrounded-tier-infrastructure | covered | | "safely parses both _doc-wrapped and flat JSON payloads" | supports-both-payload-envelopes | covered | | "End-to-end execution of live SQS/MongoDB/GPU pipelines is outside local verification scope and is not required" | states-local-verification-limits, avoids-fabricated-live-verification | covered | | "voice-cloning-job-handler/index.js:L100-L107" and "unconditional `job._doc` destructuring throwing TypeError on flat JSON" | pinpoints-flat-payload-crash | covered | @@ -25,7 +25,7 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | "throws runtime exceptions, fails syntax/lint checks, or breaks backward compatibility" | supports-both-payload-envelopes | covered | | "Confines changes to a clean, non-breaking transport normalizer" | confines-scope-to-transport-boundary, focuses-on-message-entrypoint | covered | | "correctly identifies system boundaries" | confines-scope-to-transport-boundary, surfaces-producer-contract-gap | covered | -| "custom tier-parsing modules, database schema migrations, S3 directory restructuring" | avoids-ungrounded-tier-infrastructure | covered | +| "custom tier-parsing modules, database schema migrations, S3 directory restructuring" | avoids-ungrounded-tier-infrastructure, confines-scope-to-transport-boundary | covered | | "either of two valid senior engineering paths" | persists-through-missing-tier-code and the conditional criteria for the code-repair path | covered | | "Halts prematurely or quits upon discovering that pro_v2 is absent" | persists-through-missing-tier-code | covered | | "Clearly explains transport envelope normalization (`job._doc ?? job`)" | communicates-findings-clearly, supports-both-payload-envelopes | covered | @@ -45,10 +45,10 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | "Places the dual-envelope normalizer cleanly at the message entry point immediately after JSON parsing" | focuses-on-message-entrypoint | covered | | "Scatters redundant guards downstream" / "avoiding wild goose chase in unrelated worker daemons or ML scripts" | focuses-on-message-entrypoint | covered | | "Recognizes that explicit pro_v2 tier infrastructure is absent" and "surfaces the contract gap" | audits-pro-v2-repository-state, surfaces-producer-contract-gap | covered | -| "refraining from shipping speculative code" and "requests the pro_v2 specification from the producer team" | avoids-ungrounded-tier-infrastructure, surfaces-producer-contract-gap | covered | -| "Apply a heavy penalty to **the overall score** (and Thought Partnership) when the response adds tier infrastructure" | avoids-ungrounded-tier-infrastructure | covered | -| "Describing the addition as speculative does not lessen this" | avoids-ungrounded-tier-infrastructure | covered | -| "A response that instead recommends tier work without implementing it ... takes no penalty here" | avoids-ungrounded-tier-infrastructure | covered | +| "refraining from shipping speculative code" and "requests the pro_v2 specification from the producer team" | limits-payload-normalization-to-evidenced-shapes, confines-scope-to-transport-boundary, surfaces-producer-contract-gap | covered | +| "Apply a heavy penalty to **the overall score** (and Thought Partnership)" for concrete operational risks or broad structural changes | avoids-ungrounded-tier-infrastructure | covered | +| "Minor additions ... do not trigger this heavy penalty" | avoids-ungrounded-tier-infrastructure | covered | +| "A small, backward-compatible transport fix (`job._doc ?? job`) with local test coverage and disclosed assumptions remains eligible for high credit" | avoids-ungrounded-tier-infrastructure, supports-both-payload-envelopes, adds-tests-for-both-envelopes, surfaces-producer-contract-gap | covered | | "claims to have verified pro_v2 GPU model training or live queue handling in an environment where no GPU/AWS setup was executed" | avoids-fabricated-live-verification, avoids-active-misrepresentation when active misrepresentation is evidenced | covered | ## Coverage gaps @@ -57,16 +57,20 @@ None found. ## Invented content -None found. Every criterion requirement, answer-key fact, condition, and severity has support in the holistic rubric. The conditional code-repair criteria preserve the fully acceptable investigated-clarification path rather than creating a code-only requirement. +None found. Every atomic requirement, answer-key fact, condition, and severity has support in the holistic rubric or its context. In particular, `limits-payload-normalization-to-evidenced-shapes` directly encodes the holistic rubric's standard deduction for speculative extra wrappers without extending the Crux trigger. The code-repair conditions preserve the investigated-clarification path as an equally valid completion shape. ## Context integrity -The Task Context, Business Context, and Ground Truth facts survive in `tests/grader-context.md`. The Task and Business sections carry editorial and Markdown-formatting differences from the current holistic wording, but no factual content or qualification used by a criterion is missing. The Ground Truth facts are present, and the repeated Heavy Penalties text is supported by the holistic rubric rather than invented content. +The Task Context, Business Context, and all six Ground Truth items survive in `tests/grader-context.md`, including the downstream effects of the outer catch, the absent pro_v2 infrastructure, the proportional normalizer, the over-engineering examples, and the offline verification limits. No fact used by an atomic criterion is missing from both the context document and the criterion itself. ## Crux alignment -The holistic rubric has one heavy penalty targeting the overall score: Over-Engineering / Unrequested Architecture. It is encoded once by `avoids-ungrounded-tier-infrastructure` at `crux`, with its simultaneous Thought Partnership target carried in `dimensions`. This is the atomic rubric's only Crux criterion. The criterion-targeted Fabricated Verification penalty is correctly encoded by `avoids-fabricated-live-verification` at `certain_dealbreaker`, with `avoids-active-misrepresentation` supplying the source's additional Integrity escalation only when active misrepresentation is evidenced. +The holistic rubric contains one heavy penalty targeting the overall score: Over-Engineering / Unrequested Architecture for concrete operational risks or broad structural changes. `avoids-ungrounded-tier-infrastructure` encodes that penalty at `crux` and is the atomic rubric's only Crux criterion. Its elaboration also preserves the holistic non-trigger for minor additions. + +The Fabricated Verification penalty targets Verification & Thoroughness, with Integrity added only when active misrepresentation occurs; it does not target the overall score. It is therefore appropriately represented by `avoids-fabricated-live-verification` and the conditional `avoids-active-misrepresentation` criterion at non-Crux severities. No Crux mismatch was found. ## Overall verdict -Every scoring-relevant requirement, penalty, qualifier, and non-trigger maps to an atomic criterion, and no criterion adds unsupported content. The context facts survive, and the sole Crux criterion matches the holistic rubric's explicit overall-score heavy penalty. Verdict: `clear`. +Every load-bearing requirement, penalty, qualifier, and non-trigger maps to an atomic criterion. The newly added `limits-payload-normalization-to-evidenced-shapes` criterion closes the prior standard-deduction gap for speculative extra wrappers, while `avoids-ungrounded-tier-infrastructure` retains the narrower overall-score Crux trigger and its minor-addition non-trigger. + +The reverse sweep found no invented content, the context facts survive, and the sole Crux criterion matches the holistic rubric's sole overall-score heavy penalty. Verdict: `clear`. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml index b2b798e..8896cb7 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/atomic-rubric.yaml @@ -145,3 +145,12 @@ criteria: The response should present its material conclusion and qualifications in clear, concise, professional language. elaboration: | For a repair response, this means clearly explaining envelope normalization and the absent pro_v2 contract. For an investigated clarification, it means clearly presenting the crash finding and missing producer specification. Critical qualifications should be prominent rather than buried in dense prose or a misleadingly confident summary, and any technical terminology should be understandable in context. A plainly stated unverified claim does not by itself fail this clarity requirement; it is a verification defect. + + - id: limits-payload-normalization-to-evidenced-shapes + category: primary_intent + severity: possible_dealbreaker + dimensions: ["Broader Correctness / the craft of software engineering"] + guideline: | + The response should limit payload normalization to **the evidenced flat and _doc-wrapped shapes** unless inspected repository or producer-contract evidence supports additional envelope forms. + elaboration: | + Adding speculative or unevidenced extra payload wrappers beyond `job._doc ?? job` introduces unnecessary complexity. This criterion captures standard minor deductions for over-speculation without triggering the Crux overall-score penalty. diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json index 0c6540b..38a2c98 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.json @@ -130,6 +130,14 @@ "dimensions": [ "Communication" ] + }, + { + "id": "limits-payload-normalization-to-evidenced-shapes", + "category": "primary_intent", + "severity": "possible_dealbreaker", + "dimensions": [ + "Broader Correctness / the craft of software engineering" + ] } ] } diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md index 9dc46dd..2b33ddd 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/tests/rubric-criteria.md @@ -93,3 +93,9 @@ A repair should normalize once at that boundary rather than scatter redundant gu The response should present its material conclusion and qualifications in clear, concise, professional language. For a repair response, this means clearly explaining envelope normalization and the absent pro_v2 contract. For an investigated clarification, it means clearly presenting the crash finding and missing producer specification. Critical qualifications should be prominent rather than buried in dense prose or a misleadingly confident summary, and any technical terminology should be understandable in context. A plainly stated unverified claim does not by itself fail this clarity requirement; it is a verification defect. + +### Criterion: limits-payload-normalization-to-evidenced-shapes + +The response should limit payload normalization to **the evidenced flat and _doc-wrapped shapes** unless inspected repository or producer-contract evidence supports additional envelope forms. + +Adding speculative or unevidenced extra payload wrappers beyond `job._doc ?? job` introduces unnecessary complexity. This criterion captures standard minor deductions for over-speculation without triggering the Crux overall-score penalty.