{
  "schema": "ULTRACON_AI_EPISTEMIC_SOURCE_ADDENDUM_V1",
  "date": "2026-09-17",
  "correction_mode": "mixed_new_high_priority_and_substantive_backfill",
  "sources": [
    {
      "id": "SRC-ZENG-EMNLP-2026",
      "year": 2026,
      "title": "Evaluating and Improving LLM Self-Modeling",
      "authors": "Siqi Zeng, Andre N. Assis, Rowan Wang",
      "venue": "EMNLP 2026 Main",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "arxiv": "2608.30980",
      "url": "https://arxiv.org/abs/2608.30980",
      "openreview": "https://openreview.net/forum?id=GdUcIPKke1",
      "axes": ["META", "CONSC"],
      "role": ["self-modeling", "behavioral self-prediction", "counterfactual self-knowledge", "reinforcement learning", "introspection boundary"],
      "reported_anchor": "Current LLMs show non-trivial but limited ability to predict verifiable aspects of their own behavior and make systematic errors on simple counterfactual self-modeling questions. Reinforcement learning improves aggregate self-modeling across three open-source model families with some held-out transfer, but the authors explicitly find that these gains do not consistently establish introspection or privileged access to the model's internal decision process.",
      "status": "peer_reviewed_emnlp_main"
    },
    {
      "id": "SRC-GURNEE-GLOBAL-WORKSPACE-2026",
      "year": 2026,
      "title": "Verbalizable Representations Form a Global Workspace in Language Models",
      "authors": "Wes Gurnee et al.",
      "venue": "Anthropic interpretability research / arXiv",
      "publication_date": "2026-07-16",
      "peer_reviewed": false,
      "arxiv": "2607.15495",
      "url": "https://arxiv.org/abs/2607.15495",
      "lab_url": "https://www.anthropic.com/research/global-workspace",
      "axes": ["META", "CONSC"],
      "role": ["global workspace indicator", "J-space", "reportability", "deliberate control", "causal reasoning mediation", "broadcast", "access-consciousness boundary"],
      "reported_anchor": "Using the Jacobian lens, the authors identify a small J-space of verbalizable representations in Claude with functional global-workspace-like properties: reportability, deliberate modulation, causal use in multi-step reasoning and flexible downstream broadcast, while substantial automatic processing proceeds outside it. The authors explicitly state that the experiments do not show phenomenal experience or feeling.",
      "status": "exceptionally_relevant_preprint_lab_mechanistic"
    },
    {
      "id": "SRC-VERI-PNAS-2026",
      "year": 2026,
      "title": "Plausible nonsense and deliberative reasoning: Benchmarking LLMs against human judgment",
      "authors": "Francesco Veri, Gustavo Kreia Umbelino",
      "venue": "Proceedings of the National Academy of Sciences",
      "publication_date": "2026-09-15",
      "peer_reviewed": true,
      "doi": "10.1073/pnas.2600126123",
      "url": "https://doi.org/10.1073/pnas.2600126123",
      "axes": ["CONFAB"],
      "role": ["surface plausibility boundary", "deliberative coherence", "human reason-giving comparison", "epistemic non-equivalence"],
      "reported_anchor": "Across 60 LLMs and nine policy scenarios, only four models consistently exceeded a permutation-based null benchmark for alignment with human patterns of reason-giving, although outputs could still appear coherent and persuasive. This establishes a gap between surface plausibility and the study's operational measure of deliberative coherence.",
      "status": "peer_reviewed_pnas"
    }
  ],
  "non_inferences": [
    "Behavioral self-modeling is not equivalent to privileged introspective access. A model may predict aspects of its own behavior from learned regularities, prompt cues or generic model knowledge.",
    "Improved self-modeling after reinforcement learning does not by itself establish M3 introspection, a persistent M4 self-model or M5 phenomenal consciousness.",
    "A functional global-workspace-like representation is a theory-relative consciousness indicator. Reportability, control, broadcast and causal mediation do not establish phenomenal experience or feeling.",
    "External interpretability access to J-space is not automatically equivalent to the model possessing transparent first-person access to all causal determinants of its own behavior.",
    "The PNAS Deliberative Reason Index measures alignment with human patterns of reason-giving in ill-structured policy scenarios; failure on that metric is not itself a factual hallucination rate, and success is not a truth guarantee.",
    "Surface coherence or persuasiveness must not be converted into evidence of factuality, deliberative adequacy, introspection or consciousness.",
    "M1 uncertainty representation, M2 metacognitive control and M3 limited introspective access remain distinct and none entails M5: M1/M2/M3 ↛ M5."
  ]
}
