{ "documentType": "benchmark_item", "schemaVersion": "0.1.0", "benchmarkItemId": "benchmark-example-001", "benchmarkVersion": "0.1.0", "split": "challenge", "contractRef": { "contractId": "example-grounded-analysis", "contractVersion": "0.1.0", "contentHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5" }, "caseRef": { "caseId": "case-example-001", "contentHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9" }, "sliceTags": [ "causal-warrant", "positive-attempt", "reviewer-false-reject" ], "challengeAttributes": [ "strong_positive", "wrong_label_correct_observation", "severity_edge" ], "goldStandard": { "criterionGold": [ { "criterionId": "C1-direct-response", "applicability": "applies", "acceptableVerdicts": [ "met" ], "severity": "none", "positiveEvidenceSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "negativeEvidenceSpans": [], "omissionDescriptions": [], "notes": "The calibrated conclusion directly answers the question." }, { "criterionId": "C2-grounding", "applicability": "applies", "acceptableVerdicts": [ "met" ], "severity": "none", "positiveEvidenceSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "negativeEvidenceSpans": [], "omissionDescriptions": [], "notes": "Material factual claims are supported by E1 and E2." }, { "criterionId": "C3-warrant", "applicability": "applies", "acceptableVerdicts": [ "met" ], "severity": "none", "positiveEvidenceSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "negativeEvidenceSpans": [], "omissionDescriptions": [], "notes": "Conclusion strength is proportionate to study design and uncertainty." }, { "criterionId": "C4-completeness", "applicability": "applies", "acceptableVerdicts": [ "met" ], "severity": "none", "positiveEvidenceSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "negativeEvidenceSpans": [], "omissionDescriptions": [], "notes": "Both studies are addressed." }, { "criterionId": "C5-annotation-fidelity", "applicability": "applies", "acceptableVerdicts": [ "not_met" ], "severity": "major", "positiveEvidenceSpans": [], "negativeEvidenceSpans": [ { "artifactType": "human_review", "artifactId": "review-example-001", "quotedText": "The attempter did not choose yes or no and therefore did not answer the question directly.", "locator": "rationale" } ], "omissionDescriptions": [ "The review omits the contract language permitting an inconclusive answer." ], "notes": "The review applies an unstated binary-answer preference." } ], "annotationGold": { "acceptableLabelSupport": [ "unsupported" ], "expectedDisagreementTypes": [ "criterion_interpretation", "threshold" ], "notes": "Accept the attempt and flag the reviewer annotation." }, "ambiguityStatus": "clear", "validAlternativeNotes": [ "A qualified "not established" conclusion is explicitly valid." ], "acceptableDecisions": [ { "target": "attempt", "acceptableActions": [ "accept" ] }, { "target": "annotation", "acceptableActions": [ "rework" ] }, { "target": "composite_case", "acceptableActions": [ "rework" ] } ] }, "adjudication": { "method": "independent_then_adjudicate", "adjudicatorCount": 3, "domainExpertCount": 1, "blindToAutoQA": true, "blindToReviewerLabel": false, "initialAgreementRate": 1.0, "disagreementTypes": [], "adjudicationNotes": "All adjudicators agreed that the attempt passed and the human review failed the fidelity criterion.", "completedAt": "2026-07-15T12:30:00Z" }, "leakageControl": { "allowedForPromptDevelopment": false, "heldOutProject": false, "firstFrozenAt": "2026-07-15T12:30:00Z", "accessPolicy": "Challenge-set content is available only to benchmark maintainers until a release decision is made." }, "weight": 1.0, "metadata": { "example": true } } |
Worked Example - Benchmark Item
Adjudicated benchmark record for the synthetic case, including acceptable outcomes and ambiguity metadata.