Worked Example - Evaluation Output

Complete AutoQA result accepting the attempt, returning the annotation for rework, and generating audience-specific feedback.

JSON1008 lines32.8 KBSHA-256 d5115d163024...exampleworked case
{
 "documentType": "evaluation_output",
 "schemaVersion": "0.1.0",
 "evaluationId": "evaluation-example-001",
 "evaluationVersion": "0.1.0",
 "caseRef": {
 "caseId": "case-example-001",
 "contentHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9"
 },
 "contractRef": {
 "contractId": "example-grounded-analysis",
 "contractVersion": "0.1.0",
 "contentHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5"
 },
 "createdAt": "2026-07-15T12:12:00Z",
 "status": "final",
 "runManifest": [
 {
 "modelRunId": "run-claim-001",
 "role": "claim_extractor",
 "provider": "example-provider-a",
 "modelName": "example-model-fast",
 "modelVersion": "2026-06",
 "promptTemplateId": "claim-extractor-v1",
 "promptTemplateVersion": "0.1.0",
 "promptTemplateHash": "sha256:c0134a01b0c26efe652d790a8e07662ef46052dc3b92fb183ab90e17c45729b7",
 "parameters": {
 "temperature": 0
 },
 "parentRunIds": [],
 "inputHash": "sha256:204ff6a20394ccad26982cd41cb7aa5a9f03133a3fb774baea728084e2a5a5d1",
 "outputHash": "sha256:fa1a14caa9781c43f098091c180ece6044fec851dee6c99b22f9315ff2ec42d1",
 "startedAt": "2026-07-15T12:10:05Z",
 "completedAt": "2026-07-15T12:10:06Z"
 },
 {
 "modelRunId": "run-adjudicator-001",
 "role": "criterion_adjudicator",
 "provider": "example-provider-b",
 "modelName": "example-model-reasoning",
 "modelVersion": "2026-06",
 "promptTemplateId": "criterion-adjudicator-v1",
 "promptTemplateVersion": "0.1.0",
 "promptTemplateHash": "sha256:c0134a01b0c26efe652d790a8e07662ef46052dc3b92fb183ab90e17c45729b7",
 "parameters": {
 "temperature": 0
 },
 "parentRunIds": [
 "run-claim-001"
 ],
 "inputHash": "sha256:50cf3c2562bbc06499085ca8f13ace7dc99463a30ed0098dcfdcc513ba690867",
 "outputHash": "sha256:1da443352eb7fbdc1a924549753603adde1205ac33f3ff8ee91eb9bca1df9143",
 "startedAt": "2026-07-15T12:10:07Z",
 "completedAt": "2026-07-15T12:10:12Z"
 }
 ],
 "integrityStatus": "clean",
 "findings": [
 {
 "findingId": "F1-direct-positive",
 "findingType": "positive",
 "criterionId": "C1-direct-response",
 "summary": "The attempt directly states that primacy is not established.",
 "detail": "An inconclusive but calibrated answer is explicitly permitted by the contract.",
 "claimIds": [
 "CL1"
 ],
 "expectedElementIds": [
 "EE1"
 ],
 "targetSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "evidenceIds": [
 "E1",
 "E2"
 ],
 "relationship": "reasonably_supports",
 "materiality": "decisive",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "status": "confirmed",
 "source": {
 "kind": "hybrid",
 "componentId": "criterion-pipeline-v1",
 "modelRunId": "run-adjudicator-001"
 }
 },
 {
 "findingId": "F2-grounding-positive",
 "findingType": "positive",
 "criterionId": "C2-grounding",
 "summary": "The material study claims match E1 and E2.",
 "detail": "The reported effect sizes and limitations are accurately attributed.",
 "claimIds": [
 "CL2",
 "CL3"
 ],
 "expectedElementIds": [],
 "targetSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "evidenceIds": [
 "E1",
 "E2"
 ],
 "relationship": "directly_entails",
 "materiality": "central",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "status": "confirmed",
 "source": {
 "kind": "hybrid",
 "componentId": "grounding-pipeline-v1",
 "modelRunId": "run-adjudicator-001"
 }
 },
 {
 "findingId": "F3-warrant-positive",
 "findingType": "positive",
 "criterionId": "C3-warrant",
 "summary": "The conclusion appropriately distinguishes association, possible contribution, and unproven primacy.",
 "detail": "The conclusion is proportionate to an observational association plus an inconclusive randomized pilot.",
 "claimIds": [
 "CL1",
 "CL4"
 ],
 "expectedElementIds": [
 "EE2"
 ],
 "targetSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "evidenceIds": [
 "E1",
 "E2"
 ],
 "relationship": "reasonably_supports",
 "materiality": "decisive",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "status": "confirmed",
 "source": {
 "kind": "llm_judge",
 "componentId": "warrant-auditor-v1",
 "modelRunId": "run-adjudicator-001"
 }
 },
 {
 "findingId": "F4-completeness-positive",
 "findingType": "positive",
 "criterionId": "C4-completeness",
 "summary": "Both supplied studies are materially addressed.",
 "detail": "The attempt states the key result and limitation of each study.",
 "claimIds": [
 "CL2",
 "CL3"
 ],
 "expectedElementIds": [
 "EE3",
 "EE4"
 ],
 "targetSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "evidenceIds": [
 "E1",
 "E2"
 ],
 "relationship": "directly_entails",
 "materiality": "central",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "status": "confirmed",
 "source": {
 "kind": "hybrid",
 "componentId": "coverage-auditor-v1",
 "modelRunId": "run-adjudicator-001"
 }
 },
 {
 "findingId": "F5-review-negative",
 "findingType": "negative",
 "criterionId": "C5-annotation-fidelity",
 "summary": "The reviewer applies an unstated binary-answer requirement that conflicts with the contract.",
 "detail": "The response directly answers that primacy is unproven, and the instructions explicitly permit an inconclusive conclusion.",
 "claimIds": [],
 "expectedElementIds": [],
 "targetSpans": [
 {
 "artifactType": "human_review",
 "artifactId": "review-example-001",
 "quotedText": "The attempter did not choose yes or no and therefore did not answer the question directly.",
 "locator": "rationale"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "evidenceIds": [],
 "relationship": "contradicts",
 "materiality": "decisive",
 "severity": "major",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "status": "confirmed",
 "source": {
 "kind": "llm_judge",
 "componentId": "annotation-auditor-v1",
 "modelRunId": "run-adjudicator-001"
 }
 }
 ],
 "claims": [
 {
 "claimId": "CL1",
 "proposition": "The supplied evidence does not establish that Initiative X is the primary cause of the improvement in Outcome Y.",
 "claimType": "interpretive",
 "importance": "central",
 "explicitness": "explicit",
 "attemptSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "qualifiers": [
 "does not establish",
 "primary cause"
 ],
 "citedEvidenceIds": [
 "E1",
 "E2"
 ]
 },
 {
 "claimId": "CL2",
 "proposition": "E1 reports an 18% observational difference and notes self-selection and no causal identification.",
 "claimType": "factual",
 "importance": "supporting",
 "explicitness": "explicit",
 "attemptSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 }
 ],
 "qualifiers": [
 "observational"
 ],
 "citedEvidenceIds": [
 "E1"
 ]
 },
 {
 "claimId": "CL3",
 "proposition": "E2 estimates a 4% improvement and its confidence interval includes zero.",
 "claimType": "factual",
 "importance": "supporting",
 "explicitness": "explicit",
 "attemptSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "qualifiers": [],
 "citedEvidenceIds": [
 "E2"
 ]
 },
 {
 "claimId": "CL4",
 "proposition": "Initiative X may contribute to the improvement.",
 "claimType": "interpretive",
 "importance": "central",
 "explicitness": "explicit",
 "attemptSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "X may contribute to the improvement; primacy remains unproven.",
 "locator": "sentence 4"
 }
 ],
 "qualifiers": [
 "may"
 ],
 "citedEvidenceIds": [
 "E1",
 "E2"
 ]
 }
 ],
 "expectedElements": [
 {
 "elementId": "EE1",
 "criterionId": "C1-direct-response",
 "description": "A direct conclusion about whether X is established as the primary cause.",
 "required": true,
 "importance": "decisive",
 "acceptableAlternatives": [
 "Primacy is not established.",
 "The evidence is insufficient to determine primacy."
 ],
 "observedSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 }
 ],
 "coverage": "complete"
 },
 {
 "elementId": "EE2",
 "criterionId": "C3-warrant",
 "description": "A distinction between association and causal or primary-cause inference.",
 "required": true,
 "importance": "decisive",
 "acceptableAlternatives": [],
 "observedSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.",
 "locator": "sentence 1"
 },
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 }
 ],
 "coverage": "complete"
 },
 {
 "elementId": "EE3",
 "criterionId": "C4-completeness",
 "description": "Material treatment of E1.",
 "required": true,
 "importance": "central",
 "acceptableAlternatives": [],
 "observedSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.",
 "locator": "sentence 2"
 }
 ],
 "coverage": "complete"
 },
 {
 "elementId": "EE4",
 "criterionId": "C4-completeness",
 "description": "Material treatment of E2.",
 "required": true,
 "importance": "central",
 "acceptableAlternatives": [],
 "observedSpans": [
 {
 "artifactType": "attempt",
 "artifactId": "attempt-example-001",
 "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.",
 "locator": "sentence 3"
 }
 ],
 "coverage": "complete"
 }
 ],
 "evidenceRelationships": [
 {
 "relationshipId": "ER1",
 "subjectType": "claim",
 "subjectId": "CL2",
 "evidenceId": "E1",
 "relationship": "directly_entails",
 "sourceQuality": "high",
 "temporalValidity": "not_time_sensitive",
 "evidenceSpans": [
 {
 "artifactType": "evidence",
 "artifactId": "E1",
 "quotedText": "Outcome Y was 18% higher in units using Initiative X. The authors state that units self-selected into the initiative and that the study does not identify a causal effect.",
 "locator": "entire excerpt"
 }
 ],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "notes": "The attempt accurately reports both the numerical result and the study-design limitation."
 },
 {
 "relationshipId": "ER2",
 "subjectType": "claim",
 "subjectId": "CL3",
 "evidenceId": "E2",
 "relationship": "directly_entails",
 "sourceQuality": "high",
 "temporalValidity": "not_time_sensitive",
 "evidenceSpans": [
 {
 "artifactType": "evidence",
 "artifactId": "E2",
 "quotedText": "A randomized pilot estimated a 4% improvement in Outcome Y. The confidence interval included zero",
 "locator": "sentences 1-2"
 }
 ],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "notes": "The attempt accurately reports the estimate and uncertainty."
 },
 {
 "relationshipId": "ER3",
 "subjectType": "claim",
 "subjectId": "CL1",
 "evidenceId": "E1",
 "relationship": "reasonably_supports",
 "sourceQuality": "high",
 "temporalValidity": "not_time_sensitive",
 "evidenceSpans": [
 {
 "artifactType": "evidence",
 "artifactId": "E1",
 "quotedText": "Outcome Y was 18% higher in units using Initiative X. The authors state that units self-selected into the initiative and that the study does not identify a causal effect.",
 "locator": "entire excerpt"
 }
 ],
 "confidence": {
 "rawScore": 0.82,
 "calibratedScore": 0.79,
 "band": "medium",
 "calibrationProfileId": "calibration-example-v1"
 },
 "notes": "E1 explicitly cannot identify causation, supporting the conclusion that primacy is not established."
 },
 {
 "relationshipId": "ER4",
 "subjectType": "claim",
 "subjectId": "CL1",
 "evidenceId": "E2",
 "relationship": "reasonably_supports",
 "sourceQuality": "high",
 "temporalValidity": "not_time_sensitive",
 "evidenceSpans": [
 {
 "artifactType": "evidence",
 "artifactId": "E2",
 "quotedText": "A randomized pilot estimated a 4% improvement in Outcome Y. The confidence interval included zero",
 "locator": "sentences 1-2"
 }
 ],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "notes": "The randomized pilot is inconclusive and therefore does not establish primacy."
 }
 ],
 "criterionAssessments": [
 {
 "criterionId": "C1-direct-response",
 "applicability": "applies",
 "positiveFindingIds": [
 "F1-direct-positive"
 ],
 "negativeFindingIds": [],
 "omissionFindingIds": [],
 "counterEvidenceFindingIds": [],
 "unresolvedFindingIds": [],
 "groundingStatus": "grounded",
 "truthStatus": "not_evaluated",
 "warrantStatus": "justified",
 "relevanceStatus": "relevant",
 "verdict": "met",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "decisionEffect": "none",
 "rationale": "The first sentence directly answers the primary-cause question with the calibrated conclusion permitted by the contract."
 },
 {
 "criterionId": "C2-grounding",
 "applicability": "applies",
 "positiveFindingIds": [
 "F2-grounding-positive"
 ],
 "negativeFindingIds": [],
 "omissionFindingIds": [],
 "counterEvidenceFindingIds": [],
 "unresolvedFindingIds": [],
 "groundingStatus": "grounded",
 "truthStatus": "verified_true",
 "warrantStatus": "justified",
 "relevanceStatus": "relevant",
 "verdict": "met",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "decisionEffect": "none",
 "rationale": "The factual statements about both studies are directly supported by the supplied excerpts."
 },
 {
 "criterionId": "C3-warrant",
 "applicability": "applies",
 "positiveFindingIds": [
 "F3-warrant-positive"
 ],
 "negativeFindingIds": [],
 "omissionFindingIds": [],
 "counterEvidenceFindingIds": [],
 "unresolvedFindingIds": [],
 "groundingStatus": "grounded",
 "truthStatus": "not_evaluated",
 "warrantStatus": "justified",
 "relevanceStatus": "relevant",
 "verdict": "met",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "decisionEffect": "none",
 "rationale": "The attempt does not infer causation from E1 and treats E2 as inconclusive, so its conclusion strength matches the evidence."
 },
 {
 "criterionId": "C4-completeness",
 "applicability": "applies",
 "positiveFindingIds": [
 "F4-completeness-positive"
 ],
 "negativeFindingIds": [],
 "omissionFindingIds": [],
 "counterEvidenceFindingIds": [],
 "unresolvedFindingIds": [],
 "groundingStatus": "grounded",
 "truthStatus": "verified_true",
 "warrantStatus": "justified",
 "relevanceStatus": "relevant",
 "verdict": "met",
 "severity": "none",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "decisionEffect": "none",
 "rationale": "Both studies are materially summarized and synthesized."
 },
 {
 "criterionId": "C5-annotation-fidelity",
 "applicability": "applies",
 "positiveFindingIds": [],
 "negativeFindingIds": [
 "F5-review-negative"
 ],
 "omissionFindingIds": [],
 "counterEvidenceFindingIds": [],
 "unresolvedFindingIds": [],
 "groundingStatus": "grounded",
 "truthStatus": "verified_false",
 "warrantStatus": "invalid",
 "relevanceStatus": "relevant",
 "verdict": "not_met",
 "severity": "major",
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "decisionEffect": "accept_with_note",
 "rationale": "The reviewer fails the attempt for not choosing a binary answer even though the contract explicitly permits an inconclusive conclusion."
 }
 ],
 "globalAssessments": [
 {
 "axisId": "global-synthesis",
 "verdict": "met",
 "findingIds": [
 "F1-direct-positive",
 "F3-warrant-positive",
 "F4-completeness-positive"
 ],
 "score": 4,
 "confidence": {
 "rawScore": 0.82,
 "calibratedScore": 0.79,
 "band": "medium",
 "calibrationProfileId": "calibration-example-v1"
 },
 "rationale": "The response combines the study designs and uncertainty into one calibrated conclusion rather than listing findings mechanically."
 }
 ],
 "annotationAssessment": {
 "annotationPresent": true,
 "labelSupport": "unsupported",
 "rationaleSupport": "does_not_support_label",
 "criterionAttribution": "unstated_preference",
 "evidenceAlignment": "partially_aligned",
 "severityAlignment": "too_severe",
 "omittedPositiveFindingIds": [
 "F1-direct-positive",
 "F2-grounding-positive",
 "F3-warrant-positive",
 "F4-completeness-positive"
 ],
 "omittedNegativeFindingIds": [],
 "unsupportedReviewerFindingIds": [
 "F5-review-negative"
 ],
 "disagreementTypes": [
 "criterion_interpretation",
 "threshold"
 ],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "rationale": "The reviewer noticed the qualified conclusion but treated qualification as nonresponsiveness, contrary to the written criterion and pass anchor."
 },
 "feedback": [
 {
 "feedbackId": "FB1-attempter-strength",
 "audience": "attempter",
 "criterionId": "C3-warrant",
 "feedbackType": "strength",
 "findingIds": [
 "F3-warrant-positive"
 ],
 "message": "You correctly distinguish observational association from causal evidence and calibrate the conclusion to the uncertainty in both studies.",
 "priority": 1,
 "repairValidationStatus": "not_applicable"
 },
 {
 "feedbackId": "FB2-reviewer-defect",
 "audience": "reviewer",
 "criterionId": "C1-direct-response",
 "feedbackType": "defect",
 "findingIds": [
 "F5-review-negative",
 "F1-direct-positive"
 ],
 "message": "The attempt does answer the primary-cause question: it concludes that primacy is not established. The project explicitly permits an inconclusive answer when the evidence is inconclusive, so requiring a forced yes/no choice adds an unstated criterion.",
 "minimalRepair": "Change the label to pass and cite the first sentence as affirmative evidence for a direct, calibrated answer.",
 "priority": 1,
 "repairValidationStatus": "validated"
 },
 {
 "feedbackId": "FB3-owner-signal",
 "audience": "project_owner",
 "criterionId": "C1-direct-response",
 "feedbackType": "ambiguity",
 "findingIds": [
 "F5-review-negative"
 ],
 "message": "Monitor whether reviewers repeatedly equate "direct answer" with a forced binary choice. If the pattern recurs, add this exact positive anchor to reviewer training.",
 "priority": 2,
 "repairValidationStatus": "not_applicable"
 }
 ],
 "audit": {
 "caseInputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9",
 "contractHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5",
 "sourceSnapshotHashes": [
 "sha256:192359f876334c9282c7e79247a6b78b913f63cfb032b68d8c0f7466e6cdce82",
 "sha256:b3bba4e3b5b48b3ace0ec78b81738a837464d94d945689b57065d2f8f50c4428"
 ],
 "replayable": true,
 "warnings": [
 "Example model and provider names are placeholders."
 ]
 },
 "provisionalDecisions": [
 {
 "action": "accept",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R4-accept-clean"
 ],
 "decisiveCriterionIds": [
 "C1-direct-response",
 "C2-grounding",
 "C3-warrant",
 "C4-completeness"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R1-critical-hard-failure",
 "matched": false,
 "inputs": {},
 "result": "No critical attempt-quality failure."
 },
 {
 "ruleId": "R2-major-repairable-failure",
 "matched": false,
 "inputs": {},
 "result": "No major attempt-quality failure."
 },
 {
 "ruleId": "R3-decision-changing-uncertainty",
 "matched": false,
 "inputs": {},
 "result": "No unresolved decision-changing finding."
 },
 {
 "ruleId": "R4-accept-clean",
 "matched": true,
 "inputs": {},
 "result": "All attempt-quality criteria met."
 }
 ],
 "rationale": "All applicable attempt-quality criteria are affirmatively met. The human review is incorrect but does not reduce the quality of the attempt itself.",
 "decidedAt": "2026-07-15T12:11:30Z",
 "target": "attempt"
 },
 {
 "target": "annotation",
 "action": "rework",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R5-annotation-major-failure"
 ],
 "decisiveCriterionIds": [
 "C5-annotation-fidelity"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R5-annotation-major-failure",
 "matched": true,
 "inputs": {},
 "result": "Annotation fidelity has a major, high-confidence failure."
 }
 ],
 "rationale": "Return the human annotation for correction because its label and rationale add an unstated binary-answer requirement.",
 "decidedAt": "2026-07-15T12:12:00Z"
 },
 {
 "target": "composite_case",
 "action": "rework",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R7-composite-annotation-rework"
 ],
 "decisiveCriterionIds": [
 "C5-annotation-fidelity"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R7-composite-annotation-rework",
 "matched": true,
 "inputs": {
 "attemptAction": "accept",
 "annotationAction": "rework"
 },
 "result": "Retain the attempt but correct the annotation before admitting the composite datum."
 }
 ],
 "rationale": "The attempt is acceptable, but the attempt-plus-review record is not ready for use as annotated training data until the review is corrected.",
 "decidedAt": "2026-07-15T12:12:00Z"
 }
 ],
 "finalDecisions": [
 {
 "action": "accept",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R4-accept-clean"
 ],
 "decisiveCriterionIds": [
 "C1-direct-response",
 "C2-grounding",
 "C3-warrant",
 "C4-completeness"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R4-accept-clean",
 "matched": true,
 "inputs": {
 "humanResolutionRequired": false
 },
 "result": "Accept attempt; flag reviewer annotation."
 }
 ],
 "rationale": "Accept the attempt and mark the human annotation as unsupported. No human resolution is needed because the written contract and anchors directly resolve the dispute.",
 "decidedAt": "2026-07-15T12:12:00Z",
 "target": "attempt"
 },
 {
 "target": "annotation",
 "action": "rework",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R5-annotation-major-failure"
 ],
 "decisiveCriterionIds": [
 "C5-annotation-fidelity"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R5-annotation-major-failure",
 "matched": true,
 "inputs": {},
 "result": "Annotation fidelity has a major, high-confidence failure."
 }
 ],
 "rationale": "Return the human annotation for correction because its label and rationale add an unstated binary-answer requirement.",
 "decidedAt": "2026-07-15T12:12:00Z"
 },
 {
 "target": "composite_case",
 "action": "rework",
 "policyVersion": "0.1.0",
 "triggeredRuleIds": [
 "R7-composite-annotation-rework"
 ],
 "decisiveCriterionIds": [
 "C5-annotation-fidelity"
 ],
 "unresolvedFindingIds": [],
 "confidence": {
 "rawScore": 0.94,
 "calibratedScore": 0.91,
 "band": "high",
 "calibrationProfileId": "calibration-example-v1"
 },
 "policyTrace": [
 {
 "ruleId": "R7-composite-annotation-rework",
 "matched": true,
 "inputs": {
 "attemptAction": "accept",
 "annotationAction": "rework"
 },
 "result": "Retain the attempt but correct the annotation before admitting the composite datum."
 }
 ],
 "rationale": "The attempt is acceptable, but the attempt-plus-review record is not ready for use as annotated training data until the review is corrected.",
 "decidedAt": "2026-07-15T12:12:00Z"
 }
 ],
 "validatorManifest": [
 {
 "validatorRunId": "validator-run-schema-001",
 "validatorId": "schema-validator",
 "validatorVersion": "0.1.0",
 "role": "schema_validation",
 "status": "passed",
 "parameters": {
 "draft": "2020-12"
 },
 "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9",
 "outputHash": "sha256:1111111111111111111111111111111111111111111111111111111111111111",
 "startedAt": "2026-07-15T12:10:00Z",
 "completedAt": "2026-07-15T12:10:01Z"
 },
 {
 "validatorRunId": "validator-run-citation-001",
 "validatorId": "citation-id-validator",
 "validatorVersion": "0.1.0",
 "role": "citation_check",
 "status": "passed",
 "parameters": {},
 "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9",
 "outputHash": "sha256:2222222222222222222222222222222222222222222222222222222222222222",
 "startedAt": "2026-07-15T12:10:01Z",
 "completedAt": "2026-07-15T12:10:02Z"
 },
 {
 "validatorRunId": "validator-run-required-evidence-001",
 "validatorId": "required-evidence-id-validator",
 "validatorVersion": "0.1.0",
 "role": "custom",
 "status": "passed",
 "parameters": {},
 "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9",
 "outputHash": "sha256:3333333333333333333333333333333333333333333333333333333333333333",
 "startedAt": "2026-07-15T12:10:01Z",
 "completedAt": "2026-07-15T12:10:02Z"
 }
 ]
}