{
  "paper_slug": "gadeke-2026-guilt-insula",
  "paper_doi": "10.7554/eLife.105391",
  "paper_title": "Contributions of insula and superior temporal sulcus to interpersonal guilt and responsibility in social decisions",
  "extraction_path": "jats",
  "extraction_path_note": "JATS-XML from https://cdn.elifesciences.org/articles/105391/elife-105391-v1.xml",
  "per_agent_counts": {
    "results": 44,
    "caption": 25,
    "structure": 16
  },
  "model": "supplied:runs/gadeke-2026-guilt-insula/external-review.answer.v3.json",
  "claims": [
    {
      "claim": "Responsibility for a social choice that yields a low outcome for a partner produces interpersonal guilt, experienced by the decision-maker as a larger decrease in momentary happiness than when the partner made the same choice.",
      "panel": null,
      "claim_type": "hypothesis",
      "role": "hypothesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "This study investigated the neural mechanisms involved in feelings of interpersonal guilt and responsibility evoked by social decisions in humans."
      },
      "span_by_agent": {
        "results": "abstract-001"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The paper's guiding proposition, framed via the research aim and the operational definition of guilt rather than an explicit 'we hypothesize' statement. Only the results reader could surface a hypothesis; single-source is expected.",
      "part_of": null
    },
    {
      "claim": "The anterior insula is the neural substrate of the guilt effect, increasing its BOLD response when participants are responsible for low outcomes affecting their partner.",
      "panel": null,
      "claim_type": "hypothesis",
      "role": "hypothesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Next, we sought to uncover the neural mechanisms associated with our guilt effect and those involved in tracking consequences of participants’ decisions on their partner."
      },
      "span_by_agent": {
        "results": "results-086"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The insula-as-guilt-substrate proposition the paper pursues given prior literature; stated as an aim and later as a result rather than as an explicit hypothesis.",
      "part_of": null
    },
    {
      "claim": "Functional connectivity between guilt- and responsibility-related outcome-phase regions and prefrontal cortex changes depending on whether participants decide for themselves alone or also for their partner, and on the type of choice (Safe or Risky).",
      "panel": null,
      "claim_type": "hypothesis",
      "role": "hypothesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "We hypothesized that connectivity with regions that showed guilt- and responsibility-related responses during the outcome phase (see previous paragraph) might change depending on whether participants made decisions for themselves only or for themselves and their partner, and depending on the type of choice (Safe or Risky)."
      },
      "span_by_agent": {
        "results": "results-140"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "A neural substrate tracks the participant's responsibility for the partner's outcomes: within regions sensitive to choice outcomes, the partner's reward prediction errors are represented more strongly when they arise from the participant's own choice than from the partner's choice.",
      "panel": null,
      "claim_type": "hypothesis",
      "role": "hypothesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "reviewer"
      ],
      "evidence_by_agent": {
        "reviewer": "Inferred from the paper's second stated aim ('Next, we sought to uncover the neural mechanisms associated with our guilt effect and those involved in tracking consequences of participants' decisions on their partner'), from the abstract's model-based STS finding, and from the title's pairing of the superior temporal sulcus with 'responsibility'. The draft carried the STS test (fig4h) and its interpretation but not the proposition they test."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "[reviewer] added: the paper's second organizing proposition, parallel to the insula/guilt hypothesis and named alongside it in the title and abstract - a neural substrate tracking the participant's responsibility for the partner (partner reward prediction errors arising from the participant's own choices). The results reader surfaced only the empirical STS result, not the hypothesis it tests.",
      "part_of": null
    },
    {
      "claim": "If responsibility for outcomes generates guilt, then participant happiness should decrease more after low lottery outcomes for the partner when the participant rather than the partner chose the lottery.",
      "panel": null,
      "claim_type": "prediction",
      "role": "prediction",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "In our definition, guilt occurs due to responsibility for low lottery outcomes for the partner."
      },
      "span_by_agent": {
        "results": "results-119"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The conditional deduced from the guilt hypothesis; the paper states it as an operational definition and then tests it. Kept distinct from the result that tests it (the guilt-effect interaction).",
      "part_of": null
    },
    {
      "claim": "If the anterior insula tracks guilt, then insula BOLD should be higher in the Social than the Partner condition and show a significant Social-by-low-outcome interaction.",
      "panel": null,
      "claim_type": "prediction",
      "role": "prediction",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "To identify regions likely to be involved in the guilt effect, we selected those satisfying two conditions: higher activity in the Social compared to the Partner condition, and a significant Social:LowOutcome interaction."
      },
      "span_by_agent": {
        "results": "results-124"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Phrased as the prediction the insula hypothesis commits the paper to; the text states it as the region-selection criteria.",
      "part_of": null
    },
    {
      "claim": "If responsibility for the partner's outcomes influences the participant's momentary happiness, then a computational model that includes the partner's reward prediction errors arising from the participant's own choices (social_pRPE) should explain the happiness data better than models omitting them, and social_pRPE weights should be reliably greater than zero.",
      "panel": null,
      "claim_type": "prediction",
      "role": "prediction",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "reviewer"
      ],
      "evidence_by_agent": {
        "reviewer": "Deduced from the guilt/responsibility hypothesis and the stated modelling aim ('we aimed to assess whether responsibility for these rewards, that is, taking into account whether the rewards occurred following choices made by the participant or the partner, would also influence variations in happiness'). The draft carried the tests - the Responsibility model's superior fit and the greater-than-zero social_pRPE weights - but not the prediction that motivates the computational analysis."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "[reviewer] added: the computational-route prediction the behavioural guilt/responsibility hypothesis commits the paper to, tested by the model comparison (Responsibility / Responsibility Redux best fit) and by the positive social_pRPE weights; kept distinct from the behavioural happiness-comparison prediction.",
      "part_of": null
    },
    {
      "claim": "If a neural substrate tracks the participant's responsibility for the partner's outcomes, then within regions sensitive to the outcomes of risky choices, BOLD should respond more strongly to the partner's reward prediction errors resulting from the participant's own choices than from the partner's choices.",
      "panel": null,
      "claim_type": "prediction",
      "role": "prediction",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "reviewer"
      ],
      "evidence_by_agent": {
        "reviewer": "Deduced from the responsibility-tracking hypothesis; the paper states the corresponding search directly ('we thus used this model to search for voxels responding more to partner reward prediction errors resulting from participant rather than partner choices, within the regions sensitive to outcomes of risky choices'). The draft carries the result (left STS, fig4h) but not the prediction it tests."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "[reviewer] added: the conditional the responsibility-tracking hypothesis commits the paper to, tested by the left-STS model-based result; kept distinct from that empirical result.",
      "part_of": null
    },
    {
      "claim": "If connectivity between the guilt- and responsibility-related outcome-phase regions (left insula, left STS) and prefrontal cortex depends on whether participants decide for themselves alone or also for their partner and on the type of choice, then a seed-to-voxel psychophysiological-interaction analysis seeded in these regions should reveal prefrontal clusters showing a significant Condition-by-Choice interaction.",
      "panel": null,
      "claim_type": "prediction",
      "role": "prediction",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "reviewer"
      ],
      "evidence_by_agent": {
        "reviewer": "Deduced from the connectivity hypothesis ('We hypothesized that connectivity with regions that showed guilt- and responsibility-related responses during the outcome phase might change depending on whether participants made decisions for themselves only or for themselves and their partner, and depending on the type of choice'); the paper then runs seed-to-voxel PPI analyses 'to search for connectivity changes, during the choice phase of the trial, as a function of Condition and Choice'. The draft carried the connectivity hypothesis and its tests (fig5, fig5s1) but not the prediction linking them."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "[reviewer] added: the observable the connectivity hypothesis commits the paper to; tested by the insula-IFG and STS-IFG PPI results.",
      "part_of": null
    },
    {
      "claim": "Participants' probability of choosing the risky option (lottery) increased with the difference between the expected value of the lottery and the value of the safe option (Study 1: t(4796) = 9.26, p < 3.1e–20, β = 0.074; Study 2: t(3829) = 10.62, p < 5.3e–26, β = 0.093).",
      "panel": "fig2a,fig2d",
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "As expected, participants’ probability of choosing the risky option (lottery) increased with the difference between the expected value of the lottery and the value of the safe option (Study 1: Figure 2A, t(4796) = 9.26, p < 3.1e–20, β = 0.074, 95% CI = [0.059 0.090]; Study 2: Figure 2D, t(3829) = 10.62, p < 5.3e–26, β = 0.093, 95% CI = [0.075 0.110])."
      },
      "span_by_agent": {
        "results": "results-005"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Manipulation check that choices tracked expected value as intended.",
      "part_of": null
    },
    {
      "claim": "Participants chose the risky option (lottery) more often in the Solo than the Social condition in Study 1 (t(4796) = 2.54, p = 0.011, β = 0.164) but not in Study 2 (t(3829) = 0.23, p = 0.82, β = 0.015).",
      "panel": "fig2a,fig2d",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Participants chose the lottery more often in the Solo condition than in the Social condition in Study 1 (t(4796) = 2.54, p = 0.011, β = 0.164, 95% CI = [0.038 0.291]), but this difference was not found in Study 2 (t(3829) = 0.23, p = 0.82, β = 0.015, 95% CI = [–0.109 0.138]).",
        "caption": "Participants chose the risky option slightly more often in the Solo condition than in the Social condition in Study 1 ( A ) but not in Study 2 ( D )."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "In mixed-effects regressions on choices, the Social condition significantly increased choice of the risky option in Study 1 but not in Study 2.",
      "panel": "app1table1",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "Condition Social 0.14* 0.03^ 0.01 0.01"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Row values are Study 1 probit (0.14*), Study 1 linear (0.03^), Study 2 probit (0.01), Study 2 linear (0.01). This is the regression-table counterpart to the fig2a/fig2d proportion effect, which the results reader described as Solo > Social; the sign of the 'Social' coefficient depends on the regression's reference condition, so the two are kept separate by panel rather than merged.",
      "part_of": null
    },
    {
      "claim": "There was no significant interaction between the difference in expected values and experimental conditions in either study (p > 0.52).",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "There was no significant interaction between the difference in expected values and experimental conditions in either study (p > 0.52)."
      },
      "span_by_agent": {
        "results": "results-007"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Null result.",
      "part_of": null
    },
    {
      "claim": "Risk premiums did not differ between Solo and Social conditions in either study (Study 1: t(39) = 1.53, p = 0.134, d = 0.24, BF10 = 0.49; Study 2: t(43) = –0.21, p = 0.84, d = –0.03, BF10 = 0.17).",
      "panel": "fig2b,fig2e",
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Risk premiums did not differ between Social and Solo conditions (Study 1: Figure 2B, t(39) = 1.53, p = 0.134, Cohen’s d = 0.24, BF10 = 0.49; Study 2: Figure 2E, t(43) = –0.21, p = 0.84, d = –0.03, BF10 = 0.17).",
        "caption": "( B, E ) Risk premiums did not differ between Solo and Social conditions."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Null result; evidence against social-context-driven changes in risk aversion.",
      "part_of": null
    },
    {
      "claim": "The risk-aversion parameter ρ did not differ between gain and loss trials (Study 1: t(17) = 0.21, p = 0.84, d = 0.05; Study 2: t(15) = –0.61, p = 0.55, d = 0.15), justifying pooling across gain and loss trials.",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "As ρ did not vary between gain and loss trials (Study 1: t(17) = 0.21, p = 0.84, d = 0.05; Study 2: t(15) = –0.61, p = 0.55, d = 0.15; paired t-test), we then pooled across gain and loss trials."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Null result justifying pooling across gain and loss trials.",
      "part_of": null
    },
    {
      "claim": "Participants were slightly more risk averse (higher ρ) in the Social than the Solo condition in Study 1 (t(39) = 2.27, p = 0.03, d = 0.36, BF10 = 1.69) but not in Study 2 (t(43) = 1.40, p = 0.17, d = 0.21, BF10 = 0.41).",
      "panel": "fig2c,fig2f",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "We found that participants were slightly more risk averse in the Social than in the Solo condition in Study 1 (Figure 2C, t(39) = 2.27, p = 0.03, d = 0.36, BF10 = 1.69) but not in Study 2 (Figure 2F, t(43) = 1.40, p = 0.17, d = 0.21, BF10 = 0.41).",
        "caption": "Values of the risk aversion parameter ρ in the Solo and Social conditions were broadly consistent with Risk premium values, but showed that participants were slightly more risk averse in the Social than in the Solo condition in Study 1 only (see Results)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "Participants showed very similar risk preferences whether deciding only for themselves (Solo) or for themselves and their partner (Social), with only a tendency toward higher risk aversion in the Social condition in Study 1.",
      "panel": null,
      "claim_type": "synthesis",
      "role": "synthesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "In sum, participants showed very similar risk preferences when making decisions affecting only themselves (Solo condition) or themselves and their partner (Social condition), with a tendency towards higher risk aversion in the Social condition in Study 1."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Integrates the choice, risk-premium and ρ results across both studies; synthesis is expected to be single-source (results reader only).",
      "part_of": null
    },
    {
      "claim": "Participant momentary happiness varied with the rewards the participant received in the current trial.",
      "panel": "fig3a,fig3e",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Across all trials, in both studies, participant momentary happiness correlated with rewards obtained in the current trial by the participant and by the partner.",
        "caption": "Happiness varied with rewards received by the participant ( A, E ) and by the partner ( B, F )."
      },
      "span_by_agent": {
        "results": "results-029"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The results reader stated the participant- and partner-reward correlations jointly; the caption reader anchored the participant-reward correlation to fig3a/fig3e specifically, so it is split from the partner-reward claim by panel.",
      "part_of": null
    },
    {
      "claim": "Participant momentary happiness varied with the rewards the partner received in the current trial.",
      "panel": "fig3b,fig3f",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Across all trials, in both studies, participant momentary happiness correlated with rewards obtained in the current trial by the participant and by the partner.",
        "caption": "Happiness varied with rewards received by the participant ( A, E ) and by the partner ( B, F )."
      },
      "span_by_agent": {
        "results": "results-029"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The results reader stated the participant- and partner-reward correlations jointly; the caption reader anchored the partner-reward correlation to fig3b/fig3f specifically, so it is split from the participant-reward claim by panel.",
      "part_of": null
    },
    {
      "claim": "Rutledge and colleagues established that changes in momentary happiness during a probabilistic reward task are explained by recent reward expectations and the prediction errors arising from them.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "literature-context",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Following Rutledge and colleagues’ methodology, which considers that changes in momentary happiness in response to outcomes of a probabilistic reward task are explained by the combined influence of recent reward expectations and prediction errors arising from those expectations, we fitted computational models to each participant’s happiness data."
      },
      "span_by_agent": {
        "results": "results-046"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Prior-work premise the happiness-modelling approach inherits.",
      "part_of": null
    },
    {
      "claim": "A likelihood ratio test showed the Responsibility model fitted the happiness data better than all other models, including the Responsibility Redux model (Study 1: all LR ≥ 47.36, p < 0.0001; Study 2: all LR ≥ 77.83, p < 0.0001).",
      "panel": "table1",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "a likelihood ratio test (Equation 9) revealed that the Responsibility model fitted better than all the other models, including the Responsibility Redux model (Study 1: all LR ≥47.36, p < 0.0001; Study 2: all LR ≥77.83, p < 0.0001)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Kept separate from the R² comparison to preserve the results reader's distinct verbatim quote for each statistic.",
      "part_of": null
    },
    {
      "claim": "The Responsibility model yielded higher R² values than all other models (Study 1: all t > 3.6, p < 0.007; Study 2: all t > 2.9, p < 0.034), except the Guilt-envy model in Study 1 (t = 2.19, p = 0.17).",
      "panel": "table1",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "The Responsibility model yielded higher R2 values than all the other models (Study 1: all t > 3.6, p < 0.007; Study 2: all t > 2.9, p < 0.034; Bonferroni-corrected t-tests) except for the Guilt-envy model in the data of Study 1 (t = 2.19, p = 0.17)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "Among the computational models fitted to momentary happiness data, the Responsibility Redux model achieved the best (lowest) AIC in both studies (Study 1 AIC –1499; Study 2 AIC –1195).",
      "panel": "table1",
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "Responsibility Redux 4 0.361 0.331 –999 –1499"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Best-fitting model inferred from the lowest AIC values (Study 2 Responsibility Redux AIC –1195). Note the apparent tension: the results reader instead reported the (non-Redux) Responsibility model as the best fit by likelihood-ratio test and R²; which model is 'best' depends on the metric, and the two readers anchored different models to Table 1, so they are not merged.",
      "part_of": null
    },
    {
      "claim": "The Responsibility Redux model — incorporating expected, previous and current rewards, reward prediction errors for both participant and partner, and decision-maker — predicted the variations in participants' momentary happiness well.",
      "panel": "fig3c,fig3g",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "A computational model taking into account expected, previous and current rewards, reward prediction errors for both participant and partner, and decision-maker (Responsibility Redux model, see Results) predicted the variations in participants’ momentary happiness well"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The model-based regressors used in the fMRI analyses depend on this fit.",
      "part_of": null
    },
    {
      "claim": "Momentary happiness was modelled with five computational models (Basic, Inequality, Guilt-envy, Responsibility, and Responsibility Redux) sharing separate, exponentially decaying terms for certain rewards, expected value, and reward prediction errors.",
      "panel": null,
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "structure"
      ],
      "evidence_by_agent": {
        "structure": "All models contained separate terms for certain rewards, expected value for lotteries and reward prediction errors, with influences that decayed exponentially over trials."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The Basic, Inequality and Guilt-envy models are identical to those in Rutledge et al., 2016.",
      "part_of": null
    },
    {
      "claim": "Model selection among the happiness models used likelihood-ratio tests comparing the Responsibility model pairwise against each other model, supplementing the AIC, BIC, R² and adjusted R² values.",
      "panel": null,
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "structure"
      ],
      "evidence_by_agent": {
        "structure": "we supplemented the AIC, BIC, R 2 and adjusted R 2 values reported in Table 1 with a series of likelihood ratio tests : we compared pair-wise the likelihoods of the Responsibility model given the data to the likelihoods of all the other models."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The model-comparison method that licenses treating the best-fitting model's variables as the regressors entered into the model-based fMRI GLM (GLM2). Kept distinct from the results reader's report of the likelihood-ratio outcome, which is an empirical result rather than a method.",
      "part_of": null
    },
    {
      "claim": "Participants' own reward prediction errors (sRPE) influenced happiness more than the partner's reward prediction errors (social_pRPE and partner_pRPE) (Study 1: all Z > 6.0, p < 0.001; Study 2: all Z > 3.7, p < 0.003).",
      "panel": null,
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "weights for sRPE were higher than for social_pRPE or partner_pRPE (Study 1: all Z > 6.0, p < 0.001; Study 2: all Z > 3.7, p < 0.003)."
      },
      "span_by_agent": {
        "results": "results-067"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "The partner's reward prediction errors resulting from the participants' own choices (social_pRPE) had weights greater than 0 (Responsibility model: Study 1: Z = 2.85, p = 0.004; Study 2: Z = 3.26, p = 0.001), contributing to explaining participants' momentary happiness.",
      "panel": null,
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "weights for social_pRPE were greater than 0: Responsibility model: Study 1: Z = 2.85, p = 0.004, Study 2: Z = 3.26, p = 0.001; ResponsibilityRedux model: Study 1: Z = 2.93, p = 0.003, Study 2: Z = 3.30, p = 0.001."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "A parameter-recovery procedure on synthetic data generated from each participant's estimated parameters showed the happiness-model parameters could be reliably recovered, verifying their stability.",
      "panel": "fig3s1",
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "contested",
      "sources": [
        "results",
        "caption",
        "structure"
      ],
      "evidence_by_agent": {
        "results": "The stability of these estimated parameters was verified using a parameter recovery procedure (see Methods and Figure 3—figure supplement 1).",
        "caption": "Stability of the estimated parameters of the temporal difference models was evaluated by attempting to recover parameters from synthetic data created using each participant’s real estimated parameters.",
        "structure": "The results show that the estimated parameters could be reliably recovered from noisy synthetic data."
      },
      "span_by_agent": {
        "results": "results-068"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "All three readers surfaced this and agree on the substance and panel (fig3s1), but disagree on role/type: the results and caption readers classified it as a methodological assessment (a capability warranting the model-based analysis), while the structure reader classified it as an empirical control (claim_type empirical) validating parameter stability. Resolved to methodological by majority; recorded as contested because they disagree on whether it is an assessment or a result.",
      "part_of": null
    },
    {
      "claim": "Participant happiness was lower when the participant was the decision-maker (Social + Solo vs. Partner), independent of outcome (Study 1: t(3600) = –3.92, p < 0.0001, β = –0.14; Study 2: t(2870) = –6.07, p < 0.0001, β = –0.24).",
      "panel": null,
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "we assessed whether happiness varied depending on the participant’s agency (Social + Solo vs. Partner), and found happiness to be lower when the participant chose, independent of the outcome (Study 1: t(3600) = –3.92, p < 0.0001, β = –0.14, 95% CI = [−0.20 to 0.07]; Study 2: t(2870) = –6.07, p < 0.0001, β = –0.24, 95% CI = [−0.31 to 0.16])."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The agency effect on happiness, distinct from the guilt (partner-outcome-contingent) effect.",
      "part_of": null
    },
    {
      "claim": "The lower happiness when the participant is the decision-maker may reflect responsibility aversion — a cost of the 'weight of the responsibility'.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "interpretation",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "This is interesting in itself and may reflect the drive behind responsibility aversion reported by Edelson et al.’s 2018 study: being assigned the role of the decider in a social setting may make people slightly unhappy, perhaps due to ‘weight of the responsibility’."
      },
      "span_by_agent": {
        "results": "results-073"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Interpretation of the agency effect through the responsibility-aversion literature.",
      "part_of": null
    },
    {
      "claim": "When the partner received the low lottery outcome, participant happiness was lower when the participant rather than the partner had chosen the lottery — a significant partner-outcome × decision-maker interaction (Study 1: t(1180) = 3.52, p = 0.0004, β = 0.37; Study 2: t(937) = 2.85, p = 0.0045, β = 0.33) — operationalizing interpersonal guilt.",
      "panel": "fig3d,fig3h",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Crucially, the interaction between partner outcome and decision-maker was significant (Study 1: t(1180) = 3.52, p = 0.0004, β = 0.37, 95% CI = [0.16 0.58]; Study 2: t(937) = 2.85, p = 0.0045, β = 0.33, 95% CI = [0.10 0.56]). When the partner received the low lottery outcome, participant happiness was lower when they rather than the partner had chosen the lottery (Figure 3D, H).",
        "caption": "Crucially, responsibility for low lottery outcomes for the partner decreased participant happiness more than the same outcomes following partner choices (see Results), which fits the definition of interpersonal guilt."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The core behavioural 'guilt effect'; this is the empirical result that tests the guilt prediction.",
      "part_of": null
    },
    {
      "claim": "The linear mixed model containing all three two-way interaction terms (Model 5, Equation 10) explained the happiness data significantly better than simpler models without interactions (p < 2e−5) and no worse than the model with all interactions (p > 0.5), warranting reporting the crucial partnerHigh:participantDecided (guilt) interaction from it.",
      "panel": "app1table2",
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "contested",
      "sources": [
        "caption",
        "structure"
      ],
      "evidence_by_agent": {
        "caption": "In both studies, Model 5 ( Equation 9  in the Results section of the main text), which contained all three two-way interaction terms, explained the data best, so its parameters for the crucial partnerHigh:participantDecided interaction are reported in the main text.",
        "structure": "the model reported in Equation 10 fitted the data significantly better (p < 2e−5) than the simpler models without interactions (higher total and adjusted R 2 , see Appendix 1—table 2 ), but not significantly worse than the model with all interactions (p > 0.5; tested with the ANOVA function in R)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Contested role/type: the caption reader classified this as an empirical result (Model 5 best-fitting, with its partnerHigh:participantDecided guilt coefficient significant — 0.39*** Study 1, 0.31** Study 2), while the structure reader classified it as a methodological warrant licensing the reported interaction. Resolved to methodological.",
      "part_of": null
    },
    {
      "claim": "The behavioural guilt effect (larger happiness decrease after low partner outcomes following participant rather than partner choices) is compatible with 'simple guilt'.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "interpretation",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "This behavioural effect (difference in happiness obtained when the partner received low lottery outcomes after participant rather than partner choices) is thus compatible with ‘simple guilt’, and we will thus refer to it as ‘guilt effect’."
      },
      "span_by_agent": {
        "results": "results-080"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "The guilt effect occurred whether the participant received the high lottery outcome (Study 1: t(39) = –3.58, p < 0.001, d = 0.56; Study 2: t(43) = –2.68, p = 0.01, d = 0.4) or the low outcome (Study 1: t(39) = –3.39, p = 0.002, d = 0.54; Study 2: t(43) = –3.58, p < 0.001, d = 0.54).",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "The ‘guilt effect’ occurred whether the participant received the high lottery outcome (Study 1: t(39) = –3.58, p < 0.001, d = 0.56, BF10 = 32; Study 2: t(43) = –2.68, p = 0.01, d = 0.4, BF10 = 3.8) or the low lottery outcome (Study 1: t(39) = –3.39, p = 0.002, d = 0.54, BF10 = 19; Study 2: t(43) = –3.58, p < 0.001, d = 0.54, BF10 = 33.5)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Shows the guilt effect does not depend on the participant's own outcome, strengthening (validating) the guilt interpretation.",
      "part_of": null
    },
    {
      "claim": "Responsibility for choices did not influence happiness following positive (high) lottery outcomes for the partner (both studies, all |t| < 1.3, p > 0.2, BF10 < 0.2).",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Responsibility for choices did not influence happiness following positive lottery outcomes for the partner (both studies, all |t| < 1.3, p > 0.2, BF10 < 0.2)."
      },
      "span_by_agent": {
        "results": "results-083"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Null result establishing that the guilt effect is specific to negative partner outcomes.",
      "part_of": null
    },
    {
      "claim": "In both studies, participants felt worse after low lottery outcomes for the partner when those outcomes followed their own choice rather than the partner's, which the authors interpret as interpersonal guilt.",
      "panel": null,
      "claim_type": "synthesis",
      "role": "synthesis",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Within these outcomes, participants felt worse following low lottery outcomes for the partner if those outcomes were consequences of their own choice rather than the partner’s, which we interpret as interpersonal guilt."
      },
      "span_by_agent": {
        "results": "results-085"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Integrates the guilt-effect results across both studies; synthesis is expected to be single-source (results reader only).",
      "part_of": null
    },
    {
      "claim": "The findings rest on two samples of healthy adults — Study 1 (behaviour only, N = 40) and Study 2 (fMRI, N = 44); all BOLD/fMRI results derive from Study 2, while the behavioural results come from both studies.",
      "panel": null,
      "claim_type": "assessment",
      "role": "scope",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "structure"
      ],
      "evidence_by_agent": {
        "results": "We analysed the BOLD responses of brain regions engaged during decision-making and at the time of receiving the outcomes of the choice using conventional as well as computational model-based analyses, using the fMRI data collected in Study 2.",
        "structure": "Forty healthy participants (14 male, mean age 26.1, range 22–31) participated in Study 1 (behaviour only study), and 44 healthy participants (19 male, mean (SD) age = 30.6 (6.5), range 23–50) participated in Study 2 (fMRI study)."
      },
      "span_by_agent": {
        "results": "results-087"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Global scope condition bounding the empirical claims; distinguishes the behavioural study from the fMRI study. The results reader emphasised that BOLD results come only from Study 2; the structure reader gave the two sample sizes.",
      "part_of": null
    },
    {
      "claim": "On each trial participants chose between a safe and a risky monetary option under three conditions: choosing for oneself (Solo), for oneself and the partner (Social), and having the partner choose for both (Partner).",
      "panel": null,
      "claim_type": "assessment",
      "role": "scope",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "structure"
      ],
      "evidence_by_agent": {
        "structure": "There were three kinds of trials: decisions by the participant only for themselves ( Solo condition), decisions by the participant for themselves and the partner ( Social condition), and decisions by the partner for both themselves and the participant ( Partner condition)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The within-subject responsibility manipulation on which the guilt and agency contrasts depend.",
      "part_of": null
    },
    {
      "claim": "To hold the partner's behaviour constant across participants, the partner's decisions were simulated by an algorithm that always selected the option with the highest expected value.",
      "panel": null,
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "structure"
      ],
      "evidence_by_agent": {
        "structure": "In order to ascertain constant decisions by the partner, the partner’s decisions were simulated using a simple algorithm that always selected the option with the highest expected value"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The partner was not a free agent; partner choices in the Partner condition were deterministic, which the responsibility/guilt contrasts rely on.",
      "part_of": null
    },
    {
      "claim": "Study 2 reproduced the Study 1 design inside the fMRI scanner with identical parameters except for longer inter-stimulus intervals (3–11 s) and partners who were experimenters positioned outside the scanner.",
      "panel": null,
      "claim_type": "assessment",
      "role": "scope",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "structure"
      ],
      "evidence_by_agent": {
        "structure": "In Study 2, participants performed two sessions of the experiment described above inside the fMRI scanner. All parameters were identical except that ISIs varied from 3 to 11 s (drawn randomly from a gamma distribution)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "In Study 2 the partner was experimenter MG or TW rather than another participant, so any replication of the Study 1 guilt effect holds under this changed social pairing.",
      "part_of": null
    },
    {
      "claim": "The bilateral ventral striatum was more active when participants chose the risky rather than the safe option (Cohen's d = 0.72 left, 0.85 right), irrespective of Social or Solo condition, replicating previous findings.",
      "panel": "fig4a",
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "We searched for brain regions engaged more when participants chose the risky instead of the safe option and found such responses in the bilateral ventral striatum (Cohen’s d = 0.72 and 0.85 in the left and right clusters, respectively; Figure 4A and Appendix 1—table 3), which replicates previous findings (Cui et al., 2022; Preuschoff et al., 2006).",
        "caption": "( A ) Regions showing a greater response when participants chose the risky (lottery) rather than the safe option, irrespective of Social or Solo condition."
      },
      "span_by_agent": {
        "results": "results-088"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The results reader treated this as a control replicating a known risk-related effect and validating the imaging analysis; the caption reader described it as a plain empirical result. Both are empirical measurements agreeing on panel and direction; resolved to control.",
      "part_of": null
    },
    {
      "claim": "Decisions in the Social compared with the Solo condition engaged three clusters — the precuneus (d = 0.79), left temporo-parietal junction (d = 0.59), and medial prefrontal cortex (d = 0.54).",
      "panel": "fig4b",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Three significant clusters of voxels were identified (Figure 4B and Appendix 1—table 3), in the precuneus (d = 0.79), the left temporo-parietal junction (TPJ; d = 0.59) and the medial prefrontal cortex (mPFC; d = 0.54).",
        "caption": "( B ) Regions showing a greater response when participants chose for both themselves and their partner rather than just for themselves (Social > Solo)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "Only the precuneus and TPJ showed positive Risky–Safe differences in both the Social>Solo and Social>Partner comparisons, being most active when participants chose the lottery in the Social condition.",
      "panel": "fig4c",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Only the precuneus and TPJ showed positive differences in both comparisons (Figure 4C), indicating that these regions were most active when participants chose the lottery in the Social condition, the critical situation in which participants assume responsibility over others.",
        "caption": "Coefficients of linear mixed models (LMMs) indicate that two of these regions, precuneus and TPJ, were most active when participants chose the lottery in the Social condition."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Caption states all coefficients and differences are significantly different from 0 (see Appendix 1—table 4).",
      "part_of": null
    },
    {
      "claim": "During receipt of lottery versus safe outcomes (across all conditions), clusters were more active in the bilateral anterior insula, dmPFC, right STS, bilateral ventral striatum, right dorsolateral prefrontal cortex, and bilateral inferior parietal lobe.",
      "panel": "fig4d",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "A cluster of voxels more active during receipt of lottery outcomes than outcomes of safe choices was identified in the bilateral anterior insula, dorsal mPFC (dmPFC), right superior temporal sulcus (STS), bilateral ventral striatum, right dorsolateral prefrontal cortex, and bilateral inferior parietal lobe (Figure 4D).",
        "caption": "( D ) Brain regions more active during receipt of the outcomes of lotteries than safe choices (all conditions)."
      },
      "span_by_agent": {
        "results": "results-117"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "The insula ROIs responded more to low lottery outcomes for the partner in the Social than the Partner condition — even after subtracting responses to high outcomes — mirroring the behavioural guilt effect.",
      "panel": "fig4e",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "Thus, activation in our insula ROIs increased in situations during which participants experienced guilt for low outcomes impacting their partner, compared to similar outcomes resulting from the partner’s choices.",
        "caption": "voxels here responded more to low lottery outcomes (L) for the partner when these resulted from participant’s rather than the partner’s choices, even when responses to high outcomes were subtracted (L–H)."
      },
      "span_by_agent": {
        "results": "results-127"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Caption states all coefficients and differences are significantly different from 0 (see Appendix 1—table 6).",
      "part_of": null
    },
    {
      "claim": "During the outcome phase, responses to low lottery outcomes were higher in the Social than the Partner condition in both left and right insula (InsulaL 0.41***, InsulaR 0.18***) and lower in the right middle temporal cortex (–0.12**).",
      "panel": "app1table9",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "Social 0.41*** 0.18*** –0.12**"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Columns are InsulaL (0.41***), InsulaR (0.18***) and MidTempR (–0.12**). Table-level breakdown supplementing the fig4e insula ROI result; kept separate by panel and by the added middle-temporal region.",
      "part_of": null
    },
    {
      "claim": "The difference in response between low and high lottery outcomes was greater in the Social than the Partner condition in left insula (0.44***), right insula (0.19***), and right middle temporal cortex (0.67***).",
      "panel": "app1table10",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "Social 0.44*** 0.19*** 0.67***"
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Columns are InsulaL (0.44***), InsulaR (0.19***) and MidTempR (0.67***). Table-level breakdown supplementing the fig4e insula ROI result; kept separate by panel.",
      "part_of": null
    },
    {
      "claim": "A mass-univariate voxel-wise analysis found a small left anterior insula cluster (peak T = 3.95, d = 0.59, 22 voxels) responding more to low partner outcomes following participant than partner choices, which survived small-volume family-wise-error correction (p = 0.024).",
      "panel": "fig4f",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "We found a weak response in a small cluster within the left anterior insula (peak T = 3.95, d = 0.59, 22 voxels, peak intensity at [–28 24 –4]; Figure 4F).",
        "caption": "A cluster of voxels within the left insula ROI showed higher responses to low lottery outcomes for the partner if these resulted from participant rather than partner choices."
      },
      "span_by_agent": {
        "results": "results-129"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The caption reader treated this as a control — convergent voxel-wise confirmation of the ROI-based insula guilt effect in panel E; the results reader treated it as the empirical voxel-wise guilt result. The small-volume FWE correction (p = 0.024) is reported in a following sentence by the results reader. Both are empirical measurements agreeing on panel and direction; resolved to empirical.",
      "part_of": null
    },
    {
      "claim": "Prior literature documents an association between the anterior insula and guilt.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "literature-context",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Given the documented association between anterior insula and guilt (see Introduction), we proceeded to test whether this result survived correction for family-wise errors due to multiple comparisons restricted to the left anterior insula grey matter."
      },
      "span_by_agent": {
        "results": "results-130"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Inherited premise motivating the insula small-volume correction.",
      "part_of": null
    },
    {
      "claim": "A model-based GLM (GLM2) entered the best-fitting computational (Responsibility) model's variables — certain rewards (CR), expected value (EV), participant RPE (sRPE), and partner RPE from participant choices (social_pRPE) and from partner choices (partner_pRPE) — as regressors to locate brain regions reflecting them.",
      "panel": null,
      "claim_type": "assessment",
      "role": "methodological",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "structure"
      ],
      "evidence_by_agent": {
        "results": "We used the model to create expected BOLD responses for each participant (see Methods) and as a manipulation check searched for responses in ventral striatum evoked by participant rewards (O’Doherty et al., 2004; O’Doherty et al., 2007).",
        "structure": "In addition, we created another GLM (GLM2) with regressors designed to identify brain regions whose activation reflected the variables of the best-fitting computational model (see above)."
      },
      "span_by_agent": {
        "results": "results-133"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The neural claims about tracking social_pRPE versus partner_pRPE depend on this model-based GLM being interpretable, which in turn depends on the model comparison.",
      "part_of": null
    },
    {
      "claim": "As a manipulation check, bilateral ventral striatum activation increased with expected certain rewards and the expected values of chosen lotteries, explained by a model-based regressor coding participant rewards (left: pFWE = 0.002, T = 5.63, d = 0.75; right: pFWE = 0.005, T = 5.46, d = 0.70).",
      "panel": "fig4g",
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "We found that activation in bilateral ventral striatum indeed increased with the amount of expected certain rewards and the expected values of chosen lotteries (left: pFWE = 0.002, T = 5.63, d = 0.75, Z = 5.41, 110 voxels, peak at MNI [–14 8 –8], right: pFWE = 0.005, T = 5.46, d = 0.70, Z = 5.26, 80 voxels, peak at MNI [10 10 −4]).",
        "caption": "( G ) Activation in bilateral ventral striatum explained by a computational model-based regressor coding participant rewards."
      },
      "span_by_agent": {
        "results": "results-134"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The results reader treated this as a manipulation check validating the model-based BOLD analysis; the caption reader described it as a plain empirical result. Both agree on panel and direction; resolved to control.",
      "part_of": null
    },
    {
      "claim": "One cluster in the left STS responded more to partner reward prediction errors resulting from participant rather than partner choices (pFWE = 0.022, T = 4.70, d = 0.53, 100 voxels, peak MNI [−52 –32 0]).",
      "panel": "fig4h",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "We found this effect in one cluster within the left STS (pFWE = 0.022, T = 4.70, d = 0.53, Z = 4.57, 100 voxels, peak at MNI [−52 –32 0]; Figure 4H).",
        "caption": "one cluster in the left superior temporal sulcus region showed a higher response to partner reward prediction errors resulting from participant rather than partner choices."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Caption notes this is restricted to brain regions sensitive to outcomes of risky choices.",
      "part_of": null
    },
    {
      "claim": "The authors suggest this left STS region tracks a partner's unexpected outcomes less when they do not follow from the participant's decisions.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "interpretation",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "This finding suggests that this region of the left STS tracks a partner’s unexpected outcomes less when they do not follow from the participant’s decisions."
      },
      "span_by_agent": {
        "results": "results-138"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": null,
      "part_of": null
    },
    {
      "claim": "The left superior temporal sulcus cluster responded to model-based regressors coding participant reward prediction resulting from participant and partner choices across both sessions of the experiment.",
      "panel": "fig4i",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "caption"
      ],
      "evidence_by_agent": {
        "caption": "( I ) Response in this cluster to the computational-model-based regressors coding participant reward prediction resulting from participant and partner choices, for both sessions of the experiment."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Caption describes what is plotted (coefficients with 95% confidence intervals) rather than stating a directional result; the caption reader marked it tentative.",
      "part_of": null
    },
    {
      "claim": "Prior functional connectivity work has shown network differences between social and self-only choices, midbrain–anterior cingulate interactions during guilt compensation, and links between insula connectivity and responsibility aversion.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "literature-context",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Functional connectivity analyses have revealed differences in networks engaged by social and self-only choices (Jung et al., 2013; Ogawa et al., 2018), interactions between midbrain and anterior cingulate during compensation for guilt (Yu et al., 2014), and links between insula connectivity and responsibility aversion (Edelson et al., 2018)."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Prior-work premises motivating the connectivity analysis.",
      "part_of": null
    },
    {
      "claim": "Functional connectivity between the left anterior insula (seed) and a cluster in the right inferior frontal gyrus varied with condition and choice, being highest when participants made Risky choices for themselves and Safe choices for both players (pFWE = 0.020, T = 4.34, d = 0.80, 115 voxels, peak MNI [46 16 22]).",
      "panel": "fig5",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "The first analysis revealed a cluster in the right IFG whose connectivity to the insula (the seed region) was highest when participants made Risky choices for themselves and Safe choices for both players (pFWE = 0.020, T = 4.34, d = 0.80, Z = 4.21, 115 voxels, peak at MNI [46 16 22]; Figure 5).",
        "caption": "Changes in functional connectivity between the left anterior insula (seed) and a cluster in the right inferior frontal gyrus at the time of the choice as a function of condition (Social vs. Solo) and choice (Risky or Safe)."
      },
      "span_by_agent": {
        "results": "results-142"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "The caption reader (tentative) stated the analysis but not the direction; the results reader supplied the direction and statistics.",
      "part_of": null
    },
    {
      "claim": "A left IFG cluster showed the opposite pattern of connectivity with the left STS seed — highest for Safe-self / Risky-both-players choices — but did not survive correction for multiple comparisons (p uncorrected = 0.001, T = 4.44, 35 voxels, peak MNI [–48 14 6]).",
      "panel": "fig5s1",
      "claim_type": "empirical",
      "role": "empirical",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "results",
        "caption"
      ],
      "evidence_by_agent": {
        "results": "The second analysis revealed a smaller cluster in the left IFG that did not survive corrections for multiple tests, where connectivity with the left STS (the seed region) showed the opposite pattern: connectivity was highest when participants made Safe choices for themselves and Risky choices for both players (p uncorrected = 0.001, T = 4.44, Z = 4.30, 35 voxels, peak at MNI [–48 14 6]; Figure 5—figure supplement 1).",
        "caption": "connectivity was highest when participants made Safe choices for themselves and Risky choices for both players (p uncorrected  = 0.001,  T  = 4.44,  Z  = 4.30, 35 voxels, peak at MNI [–48 14 6])."
      },
      "span_by_agent": {
        "results": "results-144"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Reported at an uncorrected threshold; did not survive multiple-comparison correction. The caption reader classified it as a control while the results reader classified it as empirical; both agree on panel and direction, resolved to empirical.",
      "part_of": null
    },
    {
      "claim": "Connectivity between the left anterior insula and the right inferior frontal gyrus varied with choice and condition, suggesting this prefrontal region is sensitive to guilt-related information during social choices.",
      "panel": null,
      "claim_type": "interpretive",
      "role": "interpretation",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "Connectivity between this region and the right inferior frontal gyrus varied depending on choice and experimental condition, suggesting that this part of prefrontal cortex is sensitive to guilt-related information during social choices."
      },
      "span_by_agent": {
        "results": "abstract-009"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Interpretation stated in the abstract.",
      "part_of": null
    },
    {
      "claim": "Dot products between individual neural guilt responses and the Yu et al. (2020) guilt-related brain signature (GRBS) were overall positive (mean = 5.22, median = 6.97, sign test p = 0.017, Cliff's Delta = 0.4), providing convergent validity with a previously published neural guilt signature.",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "The dot products between individual responses and the GRBS varied between –40.1 and 36.7, but overall these values were positive (mean = 5.22; median = 6.97; sign test: p = 0.017; Cliff’s Delta = 0.4 = medium effect size; data are not normally distributed)."
      },
      "span_by_agent": {
        "results": "results-152"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "[reviewer] role: empirical → control. The comparison against an independent, previously published neural guilt signature (Yu et al., 2020) is a convergent-validity check: its specific outcome - positive dot products - strengthens the warrant for the anterior insula as a guilt-tracking substrate rather than establishing a new primary finding, so its work in the argument is to validate the insula/guilt result. Provides convergent validity with a previously published neural guilt signature.",
      "part_of": null
    },
    {
      "claim": "Individual GRBS dot-product values did not correlate with the behavioural guilt responses (Spearman's Rho = –0.058, p = 0.725), indicating the neural signature does not track individual differences in behavioural guilt sensitivity.",
      "panel": null,
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "single-source",
      "sources": [
        "results"
      ],
      "evidence_by_agent": {
        "results": "We assessed whether inter-individual differences in these dot product values correlated with the behavioural guilt responses, but did not find a significant association [Spearman’s Rho = –0.058, p = 0.725]."
      },
      "span_by_agent": {
        "results": "results-153"
      },
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Null result: the neural signature does not track individual differences in behavioural guilt sensitivity.",
      "part_of": null
    },
    {
      "claim": "A pre-task icebreaker succeeded in establishing a positive attitude toward the partner: participants rated their partners highly (all above 8 on a 1–10 scale) on sympathy, cooperativity, honesty, openness, and sociability in both studies.",
      "panel": "app1table11",
      "claim_type": "empirical",
      "role": "control",
      "addresses": null,
      "confidence": "high",
      "sources": [
        "caption",
        "structure"
      ],
      "evidence_by_agent": {
        "caption": "How honest did they seem? 9.05 (1.11) 9.34 (1.10)",
        "structure": "participants’ average ratings of their partners in terms of sympathy, cooperativity, honesty, openness and sociability were all above 8 on a scale of 1–10, in both studies ( Appendix 1—table 11 )."
      },
      "span_by_agent": {},
      "evidence_verified": {},
      "evidence_verified_against": {},
      "notes": "Manipulation check that the social relationship was positive and non-competitive; the caption reader anchored it to Appendix 1—table 11 (ratings across the five items range roughly 8.35–9.34 across Studies 1 and 2).",
      "part_of": null
    }
  ],
  "config_snapshot": {
    "model_results": "claude-sonnet-4-6",
    "model_caption": "claude-sonnet-4-6",
    "model_structure": "claude-sonnet-4-6",
    "model_reconcile": "claude-opus-4-6",
    "prompt_variant": "default",
    "reconcile_strategy": "confidence-tagged",
    "external_review": true
  }
}
