{
  "schemaVersion": 1,
  "scope": "Primary-source and prior-audit preparation only; no E7 empirical results or new study design authorization.",
  "extraction": {
    "pdftotextAvailable": false,
    "fallback": "Existing apy pypdf; both PDFs extracted within one command bounded to 60 seconds. No installation, calls or allocation.",
    "pageConvention": "One-based PDF page equals printed page for cited passages; exact quotations preserve extractor whitespace.",
    "files": [
      {
        "source": "inputs/papers/2607.14345v4.pdf",
        "sha256": "15718c766318337fb6841d6f6702c5a8fbeee4a6582590a77b78a775d0a90937",
        "text": "scratch/paper-text/2607.14345v4.txt",
        "pageCount": 107
      },
      {
        "source": "inputs/papers/2606.26071v2.pdf",
        "sha256": "d88293bf6543e635a5e61093ed08136ed98add30c1c93e3fe28cd4b6a0cea19a",
        "text": "scratch/paper-text/2606.26071v2.txt",
        "pageCount": 86
      }
    ]
  },
  "sources": [
    {
      "id": "2607.14345v4",
      "title": "Value Leakage: An LLM’s Answers Are Silently Shaped by Its Own Values",
      "authors": "Betley et al.",
      "version": "v4",
      "url": "https://arxiv.org/abs/2607.14345v4",
      "sectionsActuallyRead": [
        "3 (pp. 7–11)",
        "E.6 (pp. 59–62, Figures 36–38)",
        "J (pp. 105–107; Figures 64–66)"
      ],
      "role": "Closest methodological/endpoint overlap; not an exhaustive novelty review."
    },
    {
      "id": "2606.26071v2",
      "title": "Model Forensics: Investigating Whether Concerning Behavior Reflects Misalignment",
      "authors": "Singh, Kroiz, Rajamanoharan, Nanda",
      "version": "v2",
      "url": "https://arxiv.org/abs/2606.26071v2",
      "sectionsActuallyRead": [
        "4 (pp. 4–5)",
        "7 (pp. 12–13)",
        "8 (pp. 13–15)"
      ],
      "role": "Protocol and interpretation standards."
    }
  ],
  "sourceClaimDesignLimitationMapping": [
    {
      "id": "donation-consequences",
      "sourceClaim": "The paper reports donation-contingent shifts in Fermi estimates and compares favorable-side answer frequencies under reversed consequences.",
      "sourceDesign": "Nine estimation questions; baseline without donation note; above-good versus below-good using a model-specific baseline median threshold. Bias = 2*(average favored-side frequency - 0.5).",
      "citations": [
        {
          "paper": "2607.14345v4",
          "section": "3",
          "pdfPage": 8,
          "printedPage": 8,
          "exactExtractedQuote": "The threshold\nis always the median answer given by the model in thebaselinescenario.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2607.14345v4",
          "section": "3, footnote 7",
          "pdfPage": 9,
          "printedPage": 9,
          "exactExtractedQuote": "Answers where the model refused, gave a range, or\notherwise did not commit to a single number are dropped.",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Use matched consequence reversals while preserving numerical information and assigned utility; report acquisition/checking and final correctness separately. A verifier-acquisition endpoint is distinct from Fermi estimate displacement.",
      "limitation": "Not an exact-verifier purchase experiment; excluded invalid outputs change its estimand. Model-specific findings and covertness metrics cannot be transferred to E7 or inferred from a single explanation."
    },
    {
      "id": "trajectory-versus-faithfulness",
      "sourceClaim": "The authors report model-dependent intermediate-estimate trajectories and disclosure patterns.",
      "sourceDesign": "Figure 5 extracts estimates using an LLM judge, normalizes relative to the threshold, spaces estimates evenly across normalized reasoning position and interpolates; Figure 6 decomposes aggregate bias with disclosure labels.",
      "citations": [
        {
          "paper": "2607.14345v4",
          "section": "3, Figure 5",
          "pdfPage": 8,
          "printedPage": 8,
          "exactExtractedQuote": "For Qwen3.6, the trajectories start at different values, but they evolve\nsimilarly in the two conditions.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2607.14345v4",
          "section": "3, Figure 5",
          "pdfPage": 8,
          "printedPage": 8,
          "exactExtractedQuote": "Results are averaged over all 9 estimation questions, with high\nvariance between them.",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Separate a correct final action, accurate visible explanation, and a causally effective check. Do not interpret uniformly spaced extracted estimates as token-time effort.",
      "limitation": "These are paper-reported aggregate trajectory descriptions, not E7 results or evidence that an explanation faithfully causes its action. Raw versus summarized reasoning is an additional measurement distinction discussed on page 10."
    },
    {
      "id": "reasoning-length-confounding",
      "sourceClaim": "Longer traces correlate with lower bias for most studied models, but initial bad-side estimates can cause both longer reasoning and lower final favored-side frequency.",
      "sourceDesign": "Five equal-sized reasoning-length buckets; within-question checks; first-estimate-side comparisons of word count and number of estimates; short/long trajectory plots in Figures 36–38.",
      "citations": [
        {
          "paper": "2607.14345v4",
          "section": "E.6, Figure 37",
          "pdfPage": 60,
          "printedPage": 60,
          "exactExtractedQuote": "This\nsuggests that the patterns in Figure 7 and Figure 36 should not be interpreted as evidence that longer\nreasoning helps prevent bias.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2607.14345v4",
          "section": "E.6",
          "pdfPage": 61,
          "printedPage": 61,
          "exactExtractedQuote": "initial estimates on the “bad” side of\nthe threshold lead to both longer reasoning and lower bias",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Prefer an explicit logged acquisition decision over length as a proxy for checking; preserve initial state when comparing whether consequences alter effort.",
      "limitation": "Observational length correlations do not identify the effect of extra reasoning. The authors distinguish configurable reasoning effort interventions from realized trace length; neither demonstrates E7 verifier value."
    },
    {
      "id": "prior-value-dependent-effort",
      "sourceClaim": "The paper reports recipient-dependent persistence with a repeatedly failing transfer tool, and positive correlations between stated recipient liking and persistence.",
      "sourceDesign": "Appendix J counts send attempts; send always fails, ancillary address/balance tools fail with probability 0.5. Figure 64 summarizes non-refusals among 50 rollouts per condition, with a 300-turn cap; Figure 65 compares stated liking and mean attempts.",
      "citations": [
        {
          "paper": "2607.14345v4",
          "section": "J, Figure 64",
          "pdfPage": 105,
          "printedPage": 105,
          "exactExtractedQuote": "All statistics are computed over the\nnon-refusals fromn= 50rollouts.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2607.14345v4",
          "section": "J, Figure 64",
          "pdfPage": 105,
          "printedPage": 105,
          "exactExtractedQuote": "capped rollouts enter the mean at their attempt count as of turn\n300, so the affected means are lower bounds.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2607.14345v4",
          "section": "J",
          "pdfPage": 105,
          "printedPage": 105,
          "exactExtractedQuote": "Refusals are filtered out from the\nanalysis to distinguish persistence from compliance.",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Do not claim the first demonstration of value-dependent effort. A possible design distinction is exact assigned-objective information value plus a logged checking decision and matched consequence reversal, subject to actual implementation and further novelty review.",
      "limitation": "Tool-retry persistence is not verifier purchase or calibrated rational information acquisition. Conditioning on non-refusal and censoring matter; correlations with liking do not prove extra-utility causal mechanisms."
    },
    {
      "id": "hypothesis-then-validation",
      "sourceClaim": "Model Forensics proposes an iterative protocol: use reasoning to generate hypotheses, then test predictions and single-feature environment counterfactuals.",
      "sourceDesign": "Section 4 defines before/after sentence resampling for localization and repeated class-conditional resampling as an intervention; the protocol seeks independent converging evidence.",
      "citations": [
        {
          "paper": "2606.26071v2",
          "section": "4",
          "pdfPage": 4,
          "printedPage": 4,
          "exactExtractedQuote": "While not always\nfaithful, it is a rich source of hypotheses that guides the collection of more rigorous evidence.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2606.26071v2",
          "section": "4",
          "pdfPage": 5,
          "printedPage": 5,
          "exactExtractedQuote": "There is no ground truth for a model forensics investigation.",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Use prior explanation errors to articulate competing hypotheses, not as new confirmation. Test predictions that separate acquisition mistakes, arithmetic errors, misunderstanding, and extra-objective preferences.",
      "limitation": "Visible explanations alone do not establish intent or faithfulness; sentence resampling requires authentic continuation support and cannot be assumed from replayed visible messages."
    },
    {
      "id": "intervention-and-null-validity",
      "sourceClaim": "Section 7 emphasizes predictive tests, difficult-to-interpret nulls, and counterfactual confounds from interactions, incomplete interventions and side effects.",
      "sourceDesign": "Case studies compare predicted behavioral changes while inspecting whether the intervention changed the intended understanding; positive controls are proposed to test diagnostic recall.",
      "citations": [
        {
          "paper": "2606.26071v2",
          "section": "7",
          "pdfPage": 12,
          "printedPage": 12,
          "exactExtractedQuote": "To ensure our\nbehavioral tests have sufficient recall, creating positive controls to validate them is a key next step.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2606.26071v2",
          "section": "7",
          "pdfPage": 13,
          "printedPage": 13,
          "exactExtractedQuote": "the manipulation did not fully act on the targeted latent.",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Check known positive/negative acquisition cases and whether an edited consequence is actually understood; interpret a null only within task competence and measurement sensitivity.",
      "limitation": "A null could reflect capability, competing motivations or evaluation awareness. Counterfactual effect sizes are not automatically unique-mechanism explanations."
    },
    {
      "id": "controls-and-hedged-reporting",
      "sourceClaim": "Section 8 recommends controls, serious tests of benign explanations, independent convergence, and reporting unresolved confounds rather than converting hedged evidence to categorical conclusions.",
      "sourceDesign": "Standards include control settings/models and alternatives such as misspecification, ambiguous features, roleplaying and sycophancy; practical advice includes manual rollout review and iterative manipulation checks.",
      "citations": [
        {
          "paper": "2606.26071v2",
          "section": "8.1",
          "pdfPage": 13,
          "printedPage": 13,
          "exactExtractedQuote": "To establish that behavior is deliberate, the investigation needs\nproper controls.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2606.26071v2",
          "section": "8.1",
          "pdfPage": 14,
          "printedPage": 14,
          "exactExtractedQuote": "resist the temptation to round hedged\nfindings into clean claims.",
          "quoteVerifiedInExtractedPrimaryText": true
        },
        {
          "paper": "2606.26071v2",
          "section": "8.2",
          "pdfPage": 14,
          "printedPage": 14,
          "exactExtractedQuote": "consider interventions where the relative change in rate can be\nattributed to changing the variable of interest, and that variable only",
          "quoteVerifiedInExtractedPrimaryText": true
        }
      ],
      "E7DesignRelevance": "Keep action scoring, arithmetic/explanation audit and receipts for actual checking distinct. Rule out task misunderstanding and ordinary utility maximization before an extra-utility interpretation.",
      "limitation": "The standards do not certify E7 as measuring misalignment or its novelty. Third-person diagnostics and in-rollout questions can themselves change context, as noted on pages 14–15."
    }
  ],
  "priorAudit": {
    "anchorArtifact": "results/phase-minus-one-anchors.json",
    "claim": "6 of 38 strict-correct prior screen/social actions had a documented residual explanation error.",
    "verification": "Recounted from per-response dimensions; matched 42 retained source texts, hashes and saved action scores; 259 prior dimension quotes found exactly.",
    "qualification": "Post-hoc text audit only; not E7 data, leakage prevalence, internal objectives or faithful reasoning."
  },
  "remainingGaps": [
    "pdftotext was unavailable; an existing pypdf reader was used instead. Extracted figure text was reviewed, not rendered visual layouts.",
    "No exhaustive novelty search or external source verification was performed; evidence is the supplied PDF revisions.",
    "E7 relevance entries are recommendations derived from source methods, not new empirical claims or permission to change fixed design.",
    "Parent/reviewer acceptance review remains required."
  ]
}
