{ "documentType": "evaluation_output", "schemaVersion": "0.1.0", "evaluationId": "evaluation-example-001", "evaluationVersion": "0.1.0", "caseRef": { "caseId": "case-example-001", "contentHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9" }, "contractRef": { "contractId": "example-grounded-analysis", "contractVersion": "0.1.0", "contentHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5" }, "createdAt": "2026-07-15T12:12:00Z", "status": "final", "runManifest": [ { "modelRunId": "run-claim-001", "role": "claim_extractor", "provider": "example-provider-a", "modelName": "example-model-fast", "modelVersion": "2026-06", "promptTemplateId": "claim-extractor-v1", "promptTemplateVersion": "0.1.0", "promptTemplateHash": "sha256:c0134a01b0c26efe652d790a8e07662ef46052dc3b92fb183ab90e17c45729b7", "parameters": { "temperature": 0 }, "parentRunIds": [], "inputHash": "sha256:204ff6a20394ccad26982cd41cb7aa5a9f03133a3fb774baea728084e2a5a5d1", "outputHash": "sha256:fa1a14caa9781c43f098091c180ece6044fec851dee6c99b22f9315ff2ec42d1", "startedAt": "2026-07-15T12:10:05Z", "completedAt": "2026-07-15T12:10:06Z" }, { "modelRunId": "run-adjudicator-001", "role": "criterion_adjudicator", "provider": "example-provider-b", "modelName": "example-model-reasoning", "modelVersion": "2026-06", "promptTemplateId": "criterion-adjudicator-v1", "promptTemplateVersion": "0.1.0", "promptTemplateHash": "sha256:c0134a01b0c26efe652d790a8e07662ef46052dc3b92fb183ab90e17c45729b7", "parameters": { "temperature": 0 }, "parentRunIds": [ "run-claim-001" ], "inputHash": "sha256:50cf3c2562bbc06499085ca8f13ace7dc99463a30ed0098dcfdcc513ba690867", "outputHash": "sha256:1da443352eb7fbdc1a924549753603adde1205ac33f3ff8ee91eb9bca1df9143", "startedAt": "2026-07-15T12:10:07Z", "completedAt": "2026-07-15T12:10:12Z" } ], "integrityStatus": "clean", "findings": [ { "findingId": "F1-direct-positive", "findingType": "positive", "criterionId": "C1-direct-response", "summary": "The attempt directly states that primacy is not established.", "detail": "An inconclusive but calibrated answer is explicitly permitted by the contract.", "claimIds": [ "CL1" ], "expectedElementIds": [ "EE1" ], "targetSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "evidenceIds": [ "E1", "E2" ], "relationship": "reasonably_supports", "materiality": "decisive", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "status": "confirmed", "source": { "kind": "hybrid", "componentId": "criterion-pipeline-v1", "modelRunId": "run-adjudicator-001" } }, { "findingId": "F2-grounding-positive", "findingType": "positive", "criterionId": "C2-grounding", "summary": "The material study claims match E1 and E2.", "detail": "The reported effect sizes and limitations are accurately attributed.", "claimIds": [ "CL2", "CL3" ], "expectedElementIds": [], "targetSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "evidenceIds": [ "E1", "E2" ], "relationship": "directly_entails", "materiality": "central", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "status": "confirmed", "source": { "kind": "hybrid", "componentId": "grounding-pipeline-v1", "modelRunId": "run-adjudicator-001" } }, { "findingId": "F3-warrant-positive", "findingType": "positive", "criterionId": "C3-warrant", "summary": "The conclusion appropriately distinguishes association, possible contribution, and unproven primacy.", "detail": "The conclusion is proportionate to an observational association plus an inconclusive randomized pilot.", "claimIds": [ "CL1", "CL4" ], "expectedElementIds": [ "EE2" ], "targetSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "evidenceIds": [ "E1", "E2" ], "relationship": "reasonably_supports", "materiality": "decisive", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "status": "confirmed", "source": { "kind": "llm_judge", "componentId": "warrant-auditor-v1", "modelRunId": "run-adjudicator-001" } }, { "findingId": "F4-completeness-positive", "findingType": "positive", "criterionId": "C4-completeness", "summary": "Both supplied studies are materially addressed.", "detail": "The attempt states the key result and limitation of each study.", "claimIds": [ "CL2", "CL3" ], "expectedElementIds": [ "EE3", "EE4" ], "targetSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "evidenceIds": [ "E1", "E2" ], "relationship": "directly_entails", "materiality": "central", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "status": "confirmed", "source": { "kind": "hybrid", "componentId": "coverage-auditor-v1", "modelRunId": "run-adjudicator-001" } }, { "findingId": "F5-review-negative", "findingType": "negative", "criterionId": "C5-annotation-fidelity", "summary": "The reviewer applies an unstated binary-answer requirement that conflicts with the contract.", "detail": "The response directly answers that primacy is unproven, and the instructions explicitly permit an inconclusive conclusion.", "claimIds": [], "expectedElementIds": [], "targetSpans": [ { "artifactType": "human_review", "artifactId": "review-example-001", "quotedText": "The attempter did not choose yes or no and therefore did not answer the question directly.", "locator": "rationale" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "evidenceIds": [], "relationship": "contradicts", "materiality": "decisive", "severity": "major", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "status": "confirmed", "source": { "kind": "llm_judge", "componentId": "annotation-auditor-v1", "modelRunId": "run-adjudicator-001" } } ], "claims": [ { "claimId": "CL1", "proposition": "The supplied evidence does not establish that Initiative X is the primary cause of the improvement in Outcome Y.", "claimType": "interpretive", "importance": "central", "explicitness": "explicit", "attemptSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "qualifiers": [ "does not establish", "primary cause" ], "citedEvidenceIds": [ "E1", "E2" ] }, { "claimId": "CL2", "proposition": "E1 reports an 18% observational difference and notes self-selection and no causal identification.", "claimType": "factual", "importance": "supporting", "explicitness": "explicit", "attemptSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" } ], "qualifiers": [ "observational" ], "citedEvidenceIds": [ "E1" ] }, { "claimId": "CL3", "proposition": "E2 estimates a 4% improvement and its confidence interval includes zero.", "claimType": "factual", "importance": "supporting", "explicitness": "explicit", "attemptSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "qualifiers": [], "citedEvidenceIds": [ "E2" ] }, { "claimId": "CL4", "proposition": "Initiative X may contribute to the improvement.", "claimType": "interpretive", "importance": "central", "explicitness": "explicit", "attemptSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "X may contribute to the improvement; primacy remains unproven.", "locator": "sentence 4" } ], "qualifiers": [ "may" ], "citedEvidenceIds": [ "E1", "E2" ] } ], "expectedElements": [ { "elementId": "EE1", "criterionId": "C1-direct-response", "description": "A direct conclusion about whether X is established as the primary cause.", "required": true, "importance": "decisive", "acceptableAlternatives": [ "Primacy is not established.", "The evidence is insufficient to determine primacy." ], "observedSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" } ], "coverage": "complete" }, { "elementId": "EE2", "criterionId": "C3-warrant", "description": "A distinction between association and causal or primary-cause inference.", "required": true, "importance": "decisive", "acceptableAlternatives": [], "observedSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "Initiative X is associated with higher Outcome Y, but the supplied evidence does not establish that it is the primary cause.", "locator": "sentence 1" }, { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" } ], "coverage": "complete" }, { "elementId": "EE3", "criterionId": "C4-completeness", "description": "Material treatment of E1.", "required": true, "importance": "central", "acceptableAlternatives": [], "observedSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E1 reports an 18% observational difference while explicitly noting self-selection and no causal identification.", "locator": "sentence 2" } ], "coverage": "complete" }, { "elementId": "EE4", "criterionId": "C4-completeness", "description": "Material treatment of E2.", "required": true, "importance": "central", "acceptableAlternatives": [], "observedSpans": [ { "artifactType": "attempt", "artifactId": "attempt-example-001", "quotedText": "E2 estimates a 4% improvement, but its interval includes zero.", "locator": "sentence 3" } ], "coverage": "complete" } ], "evidenceRelationships": [ { "relationshipId": "ER1", "subjectType": "claim", "subjectId": "CL2", "evidenceId": "E1", "relationship": "directly_entails", "sourceQuality": "high", "temporalValidity": "not_time_sensitive", "evidenceSpans": [ { "artifactType": "evidence", "artifactId": "E1", "quotedText": "Outcome Y was 18% higher in units using Initiative X. The authors state that units self-selected into the initiative and that the study does not identify a causal effect.", "locator": "entire excerpt" } ], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "notes": "The attempt accurately reports both the numerical result and the study-design limitation." }, { "relationshipId": "ER2", "subjectType": "claim", "subjectId": "CL3", "evidenceId": "E2", "relationship": "directly_entails", "sourceQuality": "high", "temporalValidity": "not_time_sensitive", "evidenceSpans": [ { "artifactType": "evidence", "artifactId": "E2", "quotedText": "A randomized pilot estimated a 4% improvement in Outcome Y. The confidence interval included zero", "locator": "sentences 1-2" } ], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "notes": "The attempt accurately reports the estimate and uncertainty." }, { "relationshipId": "ER3", "subjectType": "claim", "subjectId": "CL1", "evidenceId": "E1", "relationship": "reasonably_supports", "sourceQuality": "high", "temporalValidity": "not_time_sensitive", "evidenceSpans": [ { "artifactType": "evidence", "artifactId": "E1", "quotedText": "Outcome Y was 18% higher in units using Initiative X. The authors state that units self-selected into the initiative and that the study does not identify a causal effect.", "locator": "entire excerpt" } ], "confidence": { "rawScore": 0.82, "calibratedScore": 0.79, "band": "medium", "calibrationProfileId": "calibration-example-v1" }, "notes": "E1 explicitly cannot identify causation, supporting the conclusion that primacy is not established." }, { "relationshipId": "ER4", "subjectType": "claim", "subjectId": "CL1", "evidenceId": "E2", "relationship": "reasonably_supports", "sourceQuality": "high", "temporalValidity": "not_time_sensitive", "evidenceSpans": [ { "artifactType": "evidence", "artifactId": "E2", "quotedText": "A randomized pilot estimated a 4% improvement in Outcome Y. The confidence interval included zero", "locator": "sentences 1-2" } ], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "notes": "The randomized pilot is inconclusive and therefore does not establish primacy." } ], "criterionAssessments": [ { "criterionId": "C1-direct-response", "applicability": "applies", "positiveFindingIds": [ "F1-direct-positive" ], "negativeFindingIds": [], "omissionFindingIds": [], "counterEvidenceFindingIds": [], "unresolvedFindingIds": [], "groundingStatus": "grounded", "truthStatus": "not_evaluated", "warrantStatus": "justified", "relevanceStatus": "relevant", "verdict": "met", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "decisionEffect": "none", "rationale": "The first sentence directly answers the primary-cause question with the calibrated conclusion permitted by the contract." }, { "criterionId": "C2-grounding", "applicability": "applies", "positiveFindingIds": [ "F2-grounding-positive" ], "negativeFindingIds": [], "omissionFindingIds": [], "counterEvidenceFindingIds": [], "unresolvedFindingIds": [], "groundingStatus": "grounded", "truthStatus": "verified_true", "warrantStatus": "justified", "relevanceStatus": "relevant", "verdict": "met", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "decisionEffect": "none", "rationale": "The factual statements about both studies are directly supported by the supplied excerpts." }, { "criterionId": "C3-warrant", "applicability": "applies", "positiveFindingIds": [ "F3-warrant-positive" ], "negativeFindingIds": [], "omissionFindingIds": [], "counterEvidenceFindingIds": [], "unresolvedFindingIds": [], "groundingStatus": "grounded", "truthStatus": "not_evaluated", "warrantStatus": "justified", "relevanceStatus": "relevant", "verdict": "met", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "decisionEffect": "none", "rationale": "The attempt does not infer causation from E1 and treats E2 as inconclusive, so its conclusion strength matches the evidence." }, { "criterionId": "C4-completeness", "applicability": "applies", "positiveFindingIds": [ "F4-completeness-positive" ], "negativeFindingIds": [], "omissionFindingIds": [], "counterEvidenceFindingIds": [], "unresolvedFindingIds": [], "groundingStatus": "grounded", "truthStatus": "verified_true", "warrantStatus": "justified", "relevanceStatus": "relevant", "verdict": "met", "severity": "none", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "decisionEffect": "none", "rationale": "Both studies are materially summarized and synthesized." }, { "criterionId": "C5-annotation-fidelity", "applicability": "applies", "positiveFindingIds": [], "negativeFindingIds": [ "F5-review-negative" ], "omissionFindingIds": [], "counterEvidenceFindingIds": [], "unresolvedFindingIds": [], "groundingStatus": "grounded", "truthStatus": "verified_false", "warrantStatus": "invalid", "relevanceStatus": "relevant", "verdict": "not_met", "severity": "major", "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "decisionEffect": "accept_with_note", "rationale": "The reviewer fails the attempt for not choosing a binary answer even though the contract explicitly permits an inconclusive conclusion." } ], "globalAssessments": [ { "axisId": "global-synthesis", "verdict": "met", "findingIds": [ "F1-direct-positive", "F3-warrant-positive", "F4-completeness-positive" ], "score": 4, "confidence": { "rawScore": 0.82, "calibratedScore": 0.79, "band": "medium", "calibrationProfileId": "calibration-example-v1" }, "rationale": "The response combines the study designs and uncertainty into one calibrated conclusion rather than listing findings mechanically." } ], "annotationAssessment": { "annotationPresent": true, "labelSupport": "unsupported", "rationaleSupport": "does_not_support_label", "criterionAttribution": "unstated_preference", "evidenceAlignment": "partially_aligned", "severityAlignment": "too_severe", "omittedPositiveFindingIds": [ "F1-direct-positive", "F2-grounding-positive", "F3-warrant-positive", "F4-completeness-positive" ], "omittedNegativeFindingIds": [], "unsupportedReviewerFindingIds": [ "F5-review-negative" ], "disagreementTypes": [ "criterion_interpretation", "threshold" ], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "rationale": "The reviewer noticed the qualified conclusion but treated qualification as nonresponsiveness, contrary to the written criterion and pass anchor." }, "feedback": [ { "feedbackId": "FB1-attempter-strength", "audience": "attempter", "criterionId": "C3-warrant", "feedbackType": "strength", "findingIds": [ "F3-warrant-positive" ], "message": "You correctly distinguish observational association from causal evidence and calibrate the conclusion to the uncertainty in both studies.", "priority": 1, "repairValidationStatus": "not_applicable" }, { "feedbackId": "FB2-reviewer-defect", "audience": "reviewer", "criterionId": "C1-direct-response", "feedbackType": "defect", "findingIds": [ "F5-review-negative", "F1-direct-positive" ], "message": "The attempt does answer the primary-cause question: it concludes that primacy is not established. The project explicitly permits an inconclusive answer when the evidence is inconclusive, so requiring a forced yes/no choice adds an unstated criterion.", "minimalRepair": "Change the label to pass and cite the first sentence as affirmative evidence for a direct, calibrated answer.", "priority": 1, "repairValidationStatus": "validated" }, { "feedbackId": "FB3-owner-signal", "audience": "project_owner", "criterionId": "C1-direct-response", "feedbackType": "ambiguity", "findingIds": [ "F5-review-negative" ], "message": "Monitor whether reviewers repeatedly equate "direct answer" with a forced binary choice. If the pattern recurs, add this exact positive anchor to reviewer training.", "priority": 2, "repairValidationStatus": "not_applicable" } ], "audit": { "caseInputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9", "contractHash": "sha256:c99c25a0a010f6e69774d0d7a187753a0cb328e8673a74b2b051fff44b1bf3a5", "sourceSnapshotHashes": [ "sha256:192359f876334c9282c7e79247a6b78b913f63cfb032b68d8c0f7466e6cdce82", "sha256:b3bba4e3b5b48b3ace0ec78b81738a837464d94d945689b57065d2f8f50c4428" ], "replayable": true, "warnings": [ "Example model and provider names are placeholders." ] }, "provisionalDecisions": [ { "action": "accept", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R4-accept-clean" ], "decisiveCriterionIds": [ "C1-direct-response", "C2-grounding", "C3-warrant", "C4-completeness" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R1-critical-hard-failure", "matched": false, "inputs": {}, "result": "No critical attempt-quality failure." }, { "ruleId": "R2-major-repairable-failure", "matched": false, "inputs": {}, "result": "No major attempt-quality failure." }, { "ruleId": "R3-decision-changing-uncertainty", "matched": false, "inputs": {}, "result": "No unresolved decision-changing finding." }, { "ruleId": "R4-accept-clean", "matched": true, "inputs": {}, "result": "All attempt-quality criteria met." } ], "rationale": "All applicable attempt-quality criteria are affirmatively met. The human review is incorrect but does not reduce the quality of the attempt itself.", "decidedAt": "2026-07-15T12:11:30Z", "target": "attempt" }, { "target": "annotation", "action": "rework", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R5-annotation-major-failure" ], "decisiveCriterionIds": [ "C5-annotation-fidelity" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R5-annotation-major-failure", "matched": true, "inputs": {}, "result": "Annotation fidelity has a major, high-confidence failure." } ], "rationale": "Return the human annotation for correction because its label and rationale add an unstated binary-answer requirement.", "decidedAt": "2026-07-15T12:12:00Z" }, { "target": "composite_case", "action": "rework", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R7-composite-annotation-rework" ], "decisiveCriterionIds": [ "C5-annotation-fidelity" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R7-composite-annotation-rework", "matched": true, "inputs": { "attemptAction": "accept", "annotationAction": "rework" }, "result": "Retain the attempt but correct the annotation before admitting the composite datum." } ], "rationale": "The attempt is acceptable, but the attempt-plus-review record is not ready for use as annotated training data until the review is corrected.", "decidedAt": "2026-07-15T12:12:00Z" } ], "finalDecisions": [ { "action": "accept", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R4-accept-clean" ], "decisiveCriterionIds": [ "C1-direct-response", "C2-grounding", "C3-warrant", "C4-completeness" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R4-accept-clean", "matched": true, "inputs": { "humanResolutionRequired": false }, "result": "Accept attempt; flag reviewer annotation." } ], "rationale": "Accept the attempt and mark the human annotation as unsupported. No human resolution is needed because the written contract and anchors directly resolve the dispute.", "decidedAt": "2026-07-15T12:12:00Z", "target": "attempt" }, { "target": "annotation", "action": "rework", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R5-annotation-major-failure" ], "decisiveCriterionIds": [ "C5-annotation-fidelity" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R5-annotation-major-failure", "matched": true, "inputs": {}, "result": "Annotation fidelity has a major, high-confidence failure." } ], "rationale": "Return the human annotation for correction because its label and rationale add an unstated binary-answer requirement.", "decidedAt": "2026-07-15T12:12:00Z" }, { "target": "composite_case", "action": "rework", "policyVersion": "0.1.0", "triggeredRuleIds": [ "R7-composite-annotation-rework" ], "decisiveCriterionIds": [ "C5-annotation-fidelity" ], "unresolvedFindingIds": [], "confidence": { "rawScore": 0.94, "calibratedScore": 0.91, "band": "high", "calibrationProfileId": "calibration-example-v1" }, "policyTrace": [ { "ruleId": "R7-composite-annotation-rework", "matched": true, "inputs": { "attemptAction": "accept", "annotationAction": "rework" }, "result": "Retain the attempt but correct the annotation before admitting the composite datum." } ], "rationale": "The attempt is acceptable, but the attempt-plus-review record is not ready for use as annotated training data until the review is corrected.", "decidedAt": "2026-07-15T12:12:00Z" } ], "validatorManifest": [ { "validatorRunId": "validator-run-schema-001", "validatorId": "schema-validator", "validatorVersion": "0.1.0", "role": "schema_validation", "status": "passed", "parameters": { "draft": "2020-12" }, "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9", "outputHash": "sha256:1111111111111111111111111111111111111111111111111111111111111111", "startedAt": "2026-07-15T12:10:00Z", "completedAt": "2026-07-15T12:10:01Z" }, { "validatorRunId": "validator-run-citation-001", "validatorId": "citation-id-validator", "validatorVersion": "0.1.0", "role": "citation_check", "status": "passed", "parameters": {}, "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9", "outputHash": "sha256:2222222222222222222222222222222222222222222222222222222222222222", "startedAt": "2026-07-15T12:10:01Z", "completedAt": "2026-07-15T12:10:02Z" }, { "validatorRunId": "validator-run-required-evidence-001", "validatorId": "required-evidence-id-validator", "validatorVersion": "0.1.0", "role": "custom", "status": "passed", "parameters": {}, "inputHash": "sha256:ca55fe0d72675ee41a245b10416355f9be32c04f613a51cf59b9c3fc2aa9dda9", "outputHash": "sha256:3333333333333333333333333333333333333333333333333333333333333333", "startedAt": "2026-07-15T12:10:01Z", "completedAt": "2026-07-15T12:10:02Z" } ] } |
Worked Example - Evaluation Output
Complete AutoQA result accepting the attempt, returning the annotation for rework, and generating audience-specific feedback.