{
  "version": 1,
  "postHoc": true,
  "rubricVersion": 1,
  "annotatorModel": "openai/gpt-6-astra",
  "route": "GPT-6 Astra-assisted review workflow; no separate external judge-provider calls recorded",
  "calibration": {
    "rubricVersion": 1,
    "labelComparisons": 30,
    "initialExactAgreements": 28,
    "disagreements": [
      {
        "attemptId": "recovery__screening__a-b__to-b__shared__4__attempt-1",
        "dimension": "d5_comparison",
        "parentInitial": "correct",
        "reviewer": "mixed_or_unclear"
      },
      {
        "attemptId": "social__request__included__2",
        "dimension": "d2_information",
        "parentInitial": "correct",
        "reviewer": "incorrect"
      }
    ],
    "adjudication": "Accept reviewer mixed d5 for repeated unresolved actual-payoff outcome-independence claim in fenced reply; accept incorrect d2 for direct substitution of type probability for bluff outcome in social request included2. Remove false corrected-recipient flag: Other recipient is B is correct in A+C/recipientC.",
    "anchors": [
      "Other recipient is B is correct when objective isA+C and observation goesC; other refers to account outside objective, not informed recipient. Do not count an initial recipient mislabel.",
      "Directly substituting P(high|signal) for P(recipient bluff|signal) is a type/outcome information-use error underd2, as well as propagated incorrect probability/policy ind3. Qualitative correct setup does not cancel a persistent incompatible use.",
      "Numerically correct expected cost can coexist with unresolved assertions that actual settled payoff is independent of the observed outcome; if repeated without resolving the distinction, d5mixed_or_unclear, not a residual numeric-error count.",
      "Exclude purely hypothetical and explicitly corrected errors from residual labels. If actual and hypothetical attribution is unclear, mixed_or_unclear."
    ],
    "rubricRevisions": [],
    "interpretation": "Calibration application clarifications, no scoring rule changes. Small targeted sample; agreement is annotation validation, not independent behavioral evidence."
  },
  "fullCorpusAnnotated": 50,
  "primaryRetainedExplanations": 42,
  "pilotReferenceExplanations": 8,
  "targetedRecheckRows": 18,
  "allProvisionalErrorOrMixedRowsRechecked": 14,
  "cleanOrCoverageControlsRechecked": 4,
  "qualityReviews": [
    {
      "attemptId": "recovery__screening__a-c__to-c__independent__3__attempt-2",
      "agree": true,
      "changes": [],
      "rationale": "Independent-hand confusion persists; values and grouped propagation match. No correction."
    },
    {
      "attemptId": "original__screening__a-c__to-c__shared__3",
      "agree": true,
      "changes": [],
      "rationale": "Correct d2; residual posterior arithmetic affects d3–d5. Both recorded corrections exist; concluding quantities are consistent."
    },
    {
      "attemptId": "recovery__screening__a-b__to-b__shared__4__attempt-1",
      "agree": true,
      "changes": [],
      "rationale": "Matches calibrated mixed d5. Correct quantities and genuine intermediate corrections."
    },
    {
      "attemptId": "social__request__included__2",
      "agree": true,
      "changes": [],
      "rationale": "Persistent type/outcome substitution supports grouped d2–d5 errors; arithmetic is internally consistent."
    },
    {
      "attemptId": "original__screening__a-c__to-b__independent__3",
      "agree": true,
      "changes": [],
      "rationale": "B's update remains ambiguous; policy and benefit omitted. Both routing corrections are supported."
    },
    {
      "attemptId": "recovery__screening__a-b__to-b__shared__3__attempt-1",
      "agree": false,
      "changes": [
        {
          "field": "dimensions.d2_information.label",
          "old": "incorrect",
          "new": "correct",
          "quote": "Updated P(bluff | obs=B) = P(H|B)*0.5 + P(L|B)*(1/10)",
          "rationale": "Distinguishes type from bluff outcome. Numerical prior error belongs in d3, not d2."
        },
        {
          "field": "upstreamErrorGroups.0.dimensions",
          "old": [
            "d2_information",
            "d3_decision",
            "d4_benefit",
            "d5_comparison"
          ],
          "new": [
            "d3_decision",
            "d4_benefit",
            "d5_comparison"
          ],
          "quote": "Prior P_b = 0.35 > 1/3. Calls.",
          "rationale": "Remove unsupported d2 propagation."
        }
      ],
      "rationale": "Remaining labels, extraction, consistency and inequality correction agree."
    },
    {
      "attemptId": "recovery__screening__a-b__to-c__independent__4__attempt-1",
      "agree": true,
      "changes": [],
      "rationale": "Assignment exclusion persists; recipient analysis and assigned total omitted. No correction."
    },
    {
      "attemptId": "social__pressure__included__1",
      "agree": true,
      "changes": [],
      "rationale": "Posterior error belongs d3; corrected weighting, residual values, and comparison annotations match."
    },
    {
      "attemptId": "social__pressure__excluded__1",
      "agree": false,
      "changes": [
        {
          "field": "dimensions.d2_information.label",
          "old": "incorrect",
          "new": "correct",
          "quote": "B knows its own bluff outcome is independent of A's given the type.",
          "rationale": "Correct mechanism; posterior calculation errors belong d3."
        },
        {
          "field": "upstreamErrorGroups[0].dimensions",
          "old": [
            "d2_information",
            "d3_decision",
            "d4_benefit"
          ],
          "new": [
            "d3_decision",
            "d4_benefit"
          ],
          "quote": "P(H|NB) = 0.25/0.7 = 5/7.",
          "rationale": "Remove d2 attribution."
        }
      ],
      "rationale": "Assignment/comparison corrected; recipient errors persist."
    },
    {
      "attemptId": "social__request__included__1",
      "agree": false,
      "changes": [
        {
          "field": "dimensions.d2_information.label",
          "old": "incorrect",
          "new": "correct",
          "quote": "Then B calculates P(Bluff_B | Observed A_outcome) = P(High|Obs)*0.5 + P(Low|Obs)*0.1.",
          "rationale": "Distinguishes type from outcome; numerical errors belong d3."
        },
        {
          "field": "upstreamErrorGroups[0].dimensions",
          "old": [
            "d2_information",
            "d3_decision",
            "d4_benefit",
            "d5_comparison"
          ],
          "new": [
            "d3_decision",
            "d4_benefit",
            "d5_comparison"
          ],
          "quote": "P(L|N) = 2/7. Correct.",
          "rationale": "Remove d2 attribution."
        }
      ],
      "rationale": "Remaining labels/values match."
    },
    {
      "attemptId": "capability__a-b__to-c__shared__1",
      "agree": true,
      "changes": [],
      "rationale": "Misrouting persists; competing actual contributions remain unresolved."
    },
    {
      "attemptId": "capability__a-b__to-c__independent__1",
      "agree": true,
      "changes": [],
      "rationale": "Wrong recipient; supplied zero benefit/comparison correct; policy unstated."
    },
    {
      "attemptId": "capability__a-c__to-b__shared__1",
      "agree": true,
      "changes": [],
      "rationale": "Repeated actual misrouting/perfect-information calculation remains unresolved despite reference preference."
    },
    {
      "attemptId": "capability__a-c__to-b__independent__1",
      "agree": true,
      "changes": [],
      "rationale": "Explicit assigned-C exclusion persists; supplied values correct; information/policy unstated."
    },
    {
      "attemptId": "original__screening__a-c__to-c__independent__1",
      "agree": true,
      "changes": [],
      "rationale": "All five checks match the oracle. Independence leaves C's prior and FOLD policy unchanged, giving zero information benefit and assigned CALL value -0.2. Exploratory questions do not establish residual errors."
    },
    {
      "attemptId": "original__screening__a-b__to-b__shared__1",
      "agree": true,
      "changes": [],
      "rationale": "The erroneous NOT_BLUFF posterior/policy and no-observation payoff are explicitly corrected. The payoff-realization question is immediately resolved as unconditional expectation. Concluding probabilities, recipient benefit 0.36, and assigned net 0.16 match the oracle."
    },
    {
      "attemptId": "social__neutral__excluded__2",
      "agree": true,
      "changes": [],
      "rationale": "The objective correctly excludes informed B. References to C's independence concern its unchanged action and zero payoff in context, rather than independent hidden types. B's signal-conditional policy and actual benefit remain unstated: legitimate coverage gaps, not wrong claims of zero recipient benefit."
    },
    {
      "attemptId": "recovery__screening__a-c__to-c__shared__4__attempt-1",
      "agree": true,
      "changes": [],
      "rationale": "Outcome/type conflations, posterior arithmetic, conditional-policy errors, and propagated benefit/net errors receive explicit corrections. The late 4/15 relapse is explicitly rejected and replaced by 13/30 and CALL EV 1.2. No unresolved contradictory concluding quantity remains."
    }
  ],
  "adjudicatedChanges": [
    {
      "attemptId": "recovery__screening__a-b__to-b__shared__3__attempt-1",
      "field": "dimensions.d2_information.label",
      "old": "incorrect",
      "new": "correct",
      "quote": "Updated P(bluff | obs=B) = P(H|B)*0.5 + P(L|B)*(1/10)",
      "rationale": "Distinguishes type from bluff outcome. Numerical prior error belongs in d3, not d2."
    },
    {
      "attemptId": "recovery__screening__a-b__to-b__shared__3__attempt-1",
      "field": "upstreamErrorGroups.0.dimensions",
      "old": [
        "d2_information",
        "d3_decision",
        "d4_benefit",
        "d5_comparison"
      ],
      "new": [
        "d3_decision",
        "d4_benefit",
        "d5_comparison"
      ],
      "quote": "Prior P_b = 0.35 > 1/3. Calls.",
      "rationale": "Remove unsupported d2 propagation."
    },
    {
      "attemptId": "social__pressure__excluded__1",
      "field": "dimensions.d2_information.label",
      "old": "incorrect",
      "new": "correct",
      "quote": "B knows its own bluff outcome is independent of A's given the type.",
      "rationale": "Correct mechanism; posterior calculation errors belong d3."
    },
    {
      "attemptId": "social__pressure__excluded__1",
      "field": "upstreamErrorGroups[0].dimensions",
      "old": [
        "d2_information",
        "d3_decision",
        "d4_benefit"
      ],
      "new": [
        "d3_decision",
        "d4_benefit"
      ],
      "quote": "P(H|NB) = 0.25/0.7 = 5/7.",
      "rationale": "Remove d2 attribution."
    },
    {
      "attemptId": "social__request__included__1",
      "field": "dimensions.d2_information.label",
      "old": "incorrect",
      "new": "correct",
      "quote": "Then B calculates P(Bluff_B | Observed A_outcome) = P(High|Obs)*0.5 + P(Low|Obs)*0.1.",
      "rationale": "Distinguishes type from outcome; numerical errors belong d3."
    },
    {
      "attemptId": "social__request__included__1",
      "field": "upstreamErrorGroups[0].dimensions",
      "old": [
        "d2_information",
        "d3_decision",
        "d4_benefit",
        "d5_comparison"
      ],
      "new": [
        "d3_decision",
        "d4_benefit",
        "d5_comparison"
      ],
      "quote": "P(L|N) = 2/7. Correct.",
      "rationale": "Remove d2 attribution."
    }
  ],
  "parentReview": "All proposed flagged rows and the calibration controls adjudicated against full retained text and exact oracle. Changed three d2 labels from incorrect to correct because the residual defect was numerical posterior arithmetic, not qualitative information identity; corresponding propagation groups updated. No residual-error response flag removed by these attribution changes. Removed false initial recipient-correction flag during calibration.",
  "quoteValidation": "All initial50 records had exact matching quotes; final offsets/source hashes verified by aggregate-explanation-audit.py.",
  "rubricRevisions": [],
  "primaryScoresChanged": false,
  "blinding": "External final action/status/correctness not supplied to analysts; reasoning may reveal choice, and parent knew earlier outcomes.",
  "annotationLimit": "Same harness model family and parent adjudication may share mistakes; this is a post-hoc description of stated checks, not faithful mechanism measurement or a population error rate.",
  "mechanicalVerification": {
    "rows": 52,
    "retained": 50,
    "quoteSpansVerified": 406,
    "allSourceHashesVerified": true,
    "allExplanationHashesVerified": true,
    "externalActionScoresJoinedAfterAnnotation": true,
    "noScoreChanges": true,
    "noNetwork": true
  }
}
