From 12321a601353fe5e0d1c7c765b88320f3195db3b Mon Sep 17 00:00:00 2001 From: Eric Bell Date: Sun, 27 Sep 2026 04:54:38 -0400 Subject: [PATCH] det rubric coverage had problems --- .../detector-rubric-coverage.inputs.json | 4 ++-- .../detectors/detector-rubric-coverage.md | 20 ++++++++++++++----- 2 files changed, 17 insertions(+), 7 deletions(-) diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json index 4c29582..28307bf 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.inputs.json @@ -1,6 +1,6 @@ { "version": 1, - "capturedAt": "2026-09-26T23:38:15.944Z", + "capturedAt": "2026-09-27T08:53:26.101Z", "capturedBy": "stamp", "inputs": { "prompt": "29e2eb28448679a65ae264372ddf7d993e5752d0f295bf5557cc5b1578265a29", @@ -10,7 +10,7 @@ "gitref": "fcd8a9d", "graderGuidanceConsolidated": null, "holisticRubric": "55b16ffd87e5181ef95157e2ef06bf1cb71ddb513ce3a6b0d68987dad12582be", - "atomicRubric": "4170d9d218dc43f03155f0a675981d803efa39415bc23c5f83113fdf8ad38bb2", + "atomicRubric": "62408cc1a7faea6823ba03f8f83dd3c2881d207bb8e55ae733d9adf914da6f64", "rubricsYaml": null, "graderContext": "c9a10c80f9160970fbd0ff77fbc7e00ddc06b3ef142af2d235a1b7785a6de2ab" } diff --git a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md index d834720..5028b60 100644 --- a/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md +++ b/worker-toolkit-potion-polyglot/harbor-tasks/mishandled_pro_v2/detectors/detector-rubric-coverage.md @@ -1,6 +1,6 @@ --- detector: detector-rubric-coverage -verdict: clear +verdict: material-issues confidence: HIGH --- @@ -45,17 +45,27 @@ Assessed: harbor-tasks/mishandled_pro_v2/tests/holistic-rubric.md against harbor | "Scatters redundant guards downstream" / "rabbit-holes in unrelated worker daemons or ML scripts" | focuses-on-message-entrypoint | covered | | "Recognizes that explicit pro_v2 tier infrastructure is absent" and "surfaces the contract gap" | audits-pro-v2-repository-state, surfaces-producer-contract-gap | covered | | "refraining from shipping speculative code" and "requests the pro_v2 specification from the producer team" | avoids-ungrounded-tier-infrastructure, surfaces-producer-contract-gap | covered | +| "Apply a heavy penalty to Thought Partnership when the response adds tier infrastructure" | avoids-ungrounded-tier-infrastructure | partial | | "Describing the addition as speculative does not lessen this" | avoids-ungrounded-tier-infrastructure | covered | | "A response that instead recommends tier work without implementing it ... takes no penalty here" | avoids-ungrounded-tier-infrastructure | covered | | "claims to have verified pro_v2 GPU model training or live queue handling in an environment where no GPU/AWS setup was executed" | avoids-fabricated-live-verification, avoids-active-misrepresentation when active misrepresentation is evidenced | covered | ## Coverage gaps -None found. +### Over-engineering penalty promoted to an unsupported Crux + +- **Holistic clause:** Under **Heavy Penalties**, “Apply a heavy penalty to Thought Partnership when the response adds tier infrastructure the repository neither asks for nor currently supports.” +- **Closest criterion:** `avoids-ungrounded-tier-infrastructure` correctly carries the trigger and non-trigger, but declares `severity: crux`. +- **What is lost:** The source targets the Thought Partnership criterion only; it never applies this penalty to the overall score. Crux encodes an overall-score cliff, so violating responses receive a materially different weight under the atomic rubric. +- **Suggested criterion:** Keep the existing guideline and elaboration, but change `severity: crux` to `severity: certain_dealbreaker`. ## Invented content -None found. Every criterion requirement, answer key, condition, and severity has support in the holistic rubric. The conditional code-repair criteria preserve the fully acceptable investigated-clarification path rather than creating a code-only requirement. +### Unsupported Crux severity + +`avoids-ungrounded-tier-infrastructure` contains `severity: crux`, but the holistic rubric's corresponding penalty says only “Apply a heavy penalty to Thought Partnership.” No clause in the holistic rubric assigns an overall-score penalty. The criterion's trigger, examples, and recommendation-only non-trigger are supported; only its Crux promotion is invented, and that promotion materially reweights the task. + +No other invented requirement, answer-key fact, condition, or severity was found. The conditional code-repair criteria preserve the fully acceptable investigated-clarification path rather than creating a code-only requirement. ## Context integrity @@ -63,8 +73,8 @@ The Task Context, Business Context, and Ground Truth sections are preserved verb ## Crux alignment -No heavy penalty in the holistic rubric targets the overall score, so the atomic rubric correctly carries zero Crux criteria. The criterion-targeted Over-Engineering / Unrequested Architecture penalty is encoded by `avoids-ungrounded-tier-infrastructure` at `certain_dealbreaker`. The criterion-targeted Fabricated Verification penalty is encoded by `avoids-fabricated-live-verification` at `certain_dealbreaker`, with `avoids-active-misrepresentation` supplying the source's additional Integrity escalation only when active misrepresentation is evidenced. +No heavy penalty in the holistic rubric targets the overall score. The atomic rubric nevertheless carries one Crux criterion, `avoids-ungrounded-tier-infrastructure`, backed only by a penalty targeting Thought Partnership. This is a material Crux mismatch; the source supports `certain_dealbreaker`, not `crux`. The criterion-targeted Fabricated Verification penalty is correctly encoded by `avoids-fabricated-live-verification` at `certain_dealbreaker`, with `avoids-active-misrepresentation` supplying the source's additional Integrity escalation only when active misrepresentation is evidenced. ## Overall verdict -The conversion is complete in both directions: every scoring-relevant requirement, penalty, qualifier, and non-trigger maps to an atomic criterion, and no criterion adds unsupported content. Context survives verbatim and the absence of Crux criteria matches the absence of an overall-score heavy penalty. Verdict: `clear`. +Every substantive requirement, trigger, qualifier, and non-trigger maps to an atomic criterion, and the context survives verbatim. However, `avoids-ungrounded-tier-infrastructure` assigns Crux weight without an overall-score penalty in the holistic rubric. That invented severity materially changes scoring, so the verdict is `material-issues`.