{
  "documentType": "benchmark_item",
  "schemaVersion": "0.1.0",
  "benchmarkItemId": "benchmark-example-001",
  "benchmarkVersion": "0.1.0",
  "split": "challenge",
  "contractRef": {
    "contractId": "example-grounded-analysis",
    "contractVersion": "0.1.0",
    "contentHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5"
  },
  "caseRef": {
    "caseId": "case-example-001",
    "contentHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9"
  },
  "sliceTags": [
    "causal-warrant",
    "positive-attempt",
    "reviewer-false-reject"
  ],
  "challengeAttributes": [
    "strong_positive",
    "wrong_label_correct_observation",
    "severity_edge"
  ],
  "goldStandard": {
    "criterionGold": [
      {
        "criterionId": "C1-direct-response",
        "applicability": "applies",
        "acceptableVerdicts": [
          "met"
        ],
        "severity": "none",
        "positiveEvidenceSpans": [
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
            "locator": "sentence 1"
          }
        ],
        "negativeEvidenceSpans": [],
        "omissionDescriptions": [],
        "notes": "The calibrated conclusion directly answers the question."
      },
      {
        "criterionId": "C2-grounding",
        "applicability": "applies",
        "acceptableVerdicts": [
          "met"
        ],
        "severity": "none",
        "positiveEvidenceSpans": [
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
            "locator": "sentence 2"
          },
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
            "locator": "sentence 3"
          }
        ],
        "negativeEvidenceSpans": [],
        "omissionDescriptions": [],
        "notes": "Material factual claims are supported by E1 and E2."
      },
      {
        "criterionId": "C3-warrant",
        "applicability": "applies",
        "acceptableVerdicts": [
          "met"
        ],
        "severity": "none",
        "positiveEvidenceSpans": [
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
            "locator": "sentence 1"
          }
        ],
        "negativeEvidenceSpans": [],
        "omissionDescriptions": [],
        "notes": "Conclusion strength is proportionate to study design and uncertainty."
      },
      {
        "criterionId": "C4-completeness",
        "applicability": "applies",
        "acceptableVerdicts": [
          "met"
        ],
        "severity": "none",
        "positiveEvidenceSpans": [
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
            "locator": "sentence 2"
          },
          {
            "artifactType": "attempt",
            "artifactId": "attempt-example-001",
            "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
            "locator": "sentence 3"
          }
        ],
        "negativeEvidenceSpans": [],
        "omissionDescriptions": [],
        "notes": "Both studies are addressed."
      },
      {
        "criterionId": "C5-annotation-fidelity",
        "applicability": "applies",
        "acceptableVerdicts": [
          "not_met"
        ],
        "severity": "major",
        "positiveEvidenceSpans": [],
        "negativeEvidenceSpans": [
          {
            "artifactType": "human_review",
            "artifactId": "review-example-001",
            "quotedText": "The attempter did not choose yes or no and therefore did not answer the question directly.",
            "locator": "rationale"
          }
        ],
        "omissionDescriptions": [
          "The review omits the contract language permitting an inconclusive answer."
        ],
        "notes": "The review applies an unstated binary-answer preference."
      }
    ],
    "annotationGold": {
      "acceptableLabelSupport": [
        "unsupported"
      ],
      "expectedDisagreementTypes": [
        "criterion_interpretation",
        "threshold"
      ],
      "notes": "Accept the attempt and flag the reviewer annotation."
    },
    "ambiguityStatus": "clear",
    "validAlternativeNotes": [
      "A qualified “not established” conclusion is explicitly valid."
    ],
    "acceptableDecisions": [
      {
        "target": "attempt",
        "acceptableActions": [
          "accept"
        ]
      },
      {
        "target": "annotation",
        "acceptableActions": [
          "rework"
        ]
      },
      {
        "target": "composite_case",
        "acceptableActions": [
          "rework"
        ]
      }
    ]
  },
  "adjudication": {
    "method": "independent_then_adjudicate",
    "adjudicatorCount": 3,
    "domainExpertCount": 1,
    "blindToAutoQA": true,
    "blindToReviewerLabel": false,
    "initialAgreementRate": 1.0,
    "disagreementTypes": [],
    "adjudicationNotes": "All adjudicators agreed that the attempt passed and the human review failed the fidelity criterion.",
    "completedAt": "2026-07-15T12:30:00Z"
  },
  "leakageControl": {
    "allowedForPromptDevelopment": false,
    "heldOutProject": false,
    "firstFrozenAt": "2026-07-15T12:30:00Z",
    "accessPolicy": "Challenge-set content is available only to benchmark maintainers until a release decision is made."
  },
  "weight": 1.0,
  "metadata": {
    "example": true
  }
}
