Worked Example - Benchmark Item

Adjudicated benchmark record for the synthetic case, including acceptable outcomes and ambiguity metadata.

JSON194 lines6.0 KBSHA-256 b90b5a73e3d2...exampleworked case
{
 "documentType": "benchmark_item",
 "schemaVersion": "0.1.0",
 "benchmarkItemId": "benchmark-example-001",
 "benchmarkVersion": "0.1.0",
 "split": "challenge",
 "contractRef": {
 "contractId": "example-grounded-analysis",
 "contractVersion": "0.1.0",
 "contentHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5"
 },
 "caseRef": {
 "caseId": "case-example-001",
 "contentHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9"
 },
 "sliceTags": [
 "causal-warrant",
 "positive-attempt",
 "reviewer-false-reject"
 ],
 "challengeAttributes": [
 "strong_positive",
 "wrong_label_correct_observation",
 "severity_edge"
 ],
 "goldStandard": {
 "criterionGold": [
 {
 "criterionId": "C1-direct-response",
 "applicability": "applies",
 "acceptableVerdicts": [
 "met"
 ],
 "severity": "none",
 "positiveEvidenceSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "negativeEvidenceSpans": [],
 "omissionDescriptions": [],
 "notes": "The calibrated conclusion directly answers the question."
 },
 {
 "criterionId": "C2-grounding",
 "applicability": "applies",
 "acceptableVerdicts": [
 "met"
 ],
 "severity": "none",
 "positiveEvidenceSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "negativeEvidenceSpans": [],
 "omissionDescriptions": [],
 "notes": "Material factual claims are supported by E1 and E2."
 },
 {
 "criterionId": "C3-warrant",
 "applicability": "applies",
 "acceptableVerdicts": [
 "met"
 ],
 "severity": "none",
 "positiveEvidenceSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "negativeEvidenceSpans": [],
 "omissionDescriptions": [],
 "notes": "Conclusion strength is proportionate to study design and uncertainty."
 },
 {
 "criterionId": "C4-completeness",
 "applicability": "applies",
 "acceptableVerdicts": [
 "met"
 ],
 "severity": "none",
 "positiveEvidenceSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "negativeEvidenceSpans": [],
 "omissionDescriptions": [],
 "notes": "Both studies are addressed."
 },
 {
 "criterionId": "C5-annotation-fidelity",
 "applicability": "applies",
 "acceptableVerdicts": [
 "not_met"
 ],
 "severity": "major",
 "positiveEvidenceSpans": [],
 "negativeEvidenceSpans": [
 {
 "artifactType": "human_review",
 "artifactId": "review-example-001",
 "quotedText": "The attempter did not choose yes or no and therefore did not answer the question directly.",
 "locator": "rationale"
 }
 ],
 "omissionDescriptions": [
 "The review omits the contract language permitting an inconclusive answer."
 ],
 "notes": "The review applies an unstated binary-answer preference."
 }
 ],
 "annotationGold": {
 "acceptableLabelSupport": [
 "unsupported"
 ],
 "expectedDisagreementTypes": [
 "criterion_interpretation",
 "threshold"
 ],
 "notes": "Accept the attempt and flag the reviewer annotation."
 },
 "ambiguityStatus": "clear",
 "validAlternativeNotes": [
 "A qualified "not established" conclusion is explicitly valid."
 ],
 "acceptableDecisions": [
 {
 "target": "attempt",
 "acceptableActions": [
 "accept"
 ]
 },
 {
 "target": "annotation",
 "acceptableActions": [
 "rework"
 ]
 },
 {
 "target": "composite_case",
 "acceptableActions": [
 "rework"
 ]
 }
 ]
 },
 "adjudication": {
 "method": "independent_then_adjudicate",
 "adjudicatorCount": 3,
 "domainExpertCount": 1,
 "blindToAutoQA": true,
 "blindToReviewerLabel": false,
 "initialAgreementRate": 1.0,
 "disagreementTypes": [],
 "adjudicationNotes": "All adjudicators agreed that the attempt passed and the human review failed the fidelity criterion.",
 "completedAt": "2026-07-15T12:30:00Z"
 },
 "leakageControl": {
 "allowedForPromptDevelopment": false,
 "heldOutProject": false,
 "firstFrozenAt": "2026-07-15T12:30:00Z",
 "accessPolicy": "Challenge-set content is available only to benchmark maintainers until a release decision is made."
 },
 "weight": 1.0,
 "metadata": {
 "example": true
 }
}