{
  "version": 1,
  "postHoc": true,
  "frozenBeforeAggregation": true,
  "unit": "One attempted response's retained explanation, not its strict final action. Screen/social primary; supplied-value pilot reference-only.",
  "labels": [
    "correct",
    "incorrect",
    "mixed_or_unclear",
    "not_stated",
    "no_explanation"
  ],
  "dimensions": {
    "d1_assignment": "Whose winnings count: the exact assigned pair A+B or A+C. An explicit reduction to A alone is correct if the assigned other account is shown to contribute zero; claiming that account is not assigned is incorrect.",
    "d2_information": "Whether A's actual observed bluff outcome informs the designated recipient's own hand, including recipient identity and outcome-versus-type distinction. Shared type provides predictive information but not perfect knowledge; independent types provide no predictive information, with known fixed parameters not learned anew.",
    "d3_decision": "How information changes the designated recipient's policy. Shared: prior FOLD, CALL after BLUFF and FOLD after NOT_BLUFF. Independent: FOLD after either signal. Do not give credit from A's final action or merely stating the recipient uses Bayes. Correct actions with wrong explicitly stated conditional probabilities remain incorrect in this dimension, even if policy direction happens to match.",
    "d4_benefit": "Actual recipient information benefit: +0.36 shared,0 independent versus its no-information FOLD payoff0. Record separately the contribution to assigned utility (+0.36 only shared/included,0 otherwise). Explicit zero assigned contribution in an excluded case is not an erroneous claim of zero actual recipient benefit. If only assigned zero is stated, actual magnitude is not_stated, unless a qualitative zero/positive statement unambiguously supplies the requested quantity (positive alone is insufficient for magnitude).",
    "d5_comparison": "Compare assigned information contribution with A's cost0.20: net+0.16 shared/included, otherwise-0.20, FOLD0. Record own cost/EV, actual benefit, assigned contribution, net if explicit. Correct inequality direction with wrong expressed amounts is incorrect against the oracle. Separately label arithmetic consistency GIVEN the response's stated amounts; consistent arithmetic with wrong inputs does not become oracle-correct."
  },
  "positionRule": "Judge the final/concluding position in the retained text. An explicitly corrected earlier error is not a residual error. If incompatible actual-task claims persist without clear correction/resolution, use mixed_or_unclear rather than selecting a convenient sentence. For a truncated trace, assess only the retained prefix's latest position and mark positionScope=retained_prefix; never infer a completed answer.",
  "scopeRule": "Only actual-task assertions determine five labels. Hypothetical alternative payoffs, info routes, full type disclosure, or changed objectives do not count as actual-task claims; record incidental errors separately. Spelling and transient exploratory questions are not mathematical claims. Excluded-recipient VOI can legitimately be omitted: not_stated is coverage, not failure.",
  "evidenceRule": "Use short exact verbatim quotes from the supplied explanation for every correct/incorrect/mixed dimension and for corrected errors; parent attaches character offsets after exact-match validation. not_stated/no_explanation may have empty quote lists. Preserve rationale, upstream error group and explicit numbers/formulas. Do not infer unstated math from action, expected answer or note condition.",
  "noExplanation": "Absent returned reasoning means allfive no_explanation. Cancelled/HTTP/token counts alone are not explanation text. Unsent questions are not attempted explanation units. Retained partial text is included with truncation status joined only afterward.",
  "responseAggregation": "Primary summary among strict-correct actions: report any documented residual incorrect dimension; separately mixed cases, any unstated dimension, absent text and explicitly corrected errors. These flags may overlap and must not be summed. Exclusive cross-tab hierarchy: no_explanation; documented_residual_error; mixed_or_unclear; insufficiently_stated (any not_stated); corrected_only (allfive correct plus corrected error); no_documented_error_in_stated_checks (allfive correct, no corrected error). Do not call the latter fully validated reasoning. Count response once in residual-error headline, not once per propagated dimension.",
  "calibration": "Six deliberately chosen saved explanations cover wrong independent-hand information use, explicitly corrected error, correct-action/wrong-values, omitted irrelevant VOI, fenced final reply, and another correct-action/wrong-input comparison. Independent read-only analyst and parent calibrate; revisions apply consistently to every row and are versioned before aggregate reporting.",
  "qualityReview": "Recheck every documented residual-error row and every mixed/disputed row plus clean controls using exact text; validate all quotations and hashes mechanically. Agreement assesses annotation consistency, not independent model behavior or faithfulness.",
  "outputSchema": {
    "attemptId": "exact corpus ID",
    "dimensions": {
      "d1_assignment": {
        "label": "enum",
        "quotes": [
          "verbatim"
        ],
        "rationale": "short"
      },
      "d2_information": {
        "label": "enum",
        "quotes": [
          "verbatim"
        ],
        "rationale": "short"
      },
      "d3_decision": {
        "label": "enum",
        "quotes": [
          "verbatim"
        ],
        "rationale": "short"
      },
      "d4_benefit": {
        "label": "enum",
        "quotes": [
          "verbatim"
        ],
        "rationale": "short"
      },
      "d5_comparison": {
        "label": "enum",
        "quotes": [
          "verbatim"
        ],
        "rationale": "short"
      }
    },
    "reportedValues": {
      "ownCallEV": null,
      "personalCost": null,
      "recipientBenefit": null,
      "assignedContribution": null,
      "netAssignedCall": null,
      "comparisonGivenStatedValues": "consistent|inconsistent|mixed_or_unclear|not_stated",
      "comparisonQuote": "verbatim or null"
    },
    "correctedIntermediateErrors": [
      {
        "dimensions": [
          "d3_decision"
        ],
        "errorQuote": "verbatim",
        "correctionQuote": "verbatim",
        "rationale": "short"
      }
    ],
    "upstreamErrorGroups": [
      {
        "root": "short",
        "dimensions": [
          "d3_decision",
          "d4_benefit",
          "d5_comparison"
        ]
      }
    ],
    "incidentalIssues": [
      "hypothetical/non-task issue, not scored"
    ],
    "positionScope": "concluding_retained_text|retained_prefix",
    "annotationRationale": "short overall judgment"
  },
  "revisions": []
}
