{
"schema":"ULTRACON_AI_EPISTEMIC_SOURCE_ADDENDUM_V2","date":"2026-09-17","mode":"deep_synthesis_additive","sources":[
{"id":"SRC-FLEMING-NRN-2026","year":2026,"title":"Towards an integrative neuroscience of metacognition","authors":"Stephen M. Fleming","venue":"Nature Reviews Neuroscience","publication_date":"2026-09-16","peer_reviewed":true,"doi":"10.1038/s41583-026-01081-x","url":"https://www.nature.com/articles/s41583-026-01081-x","axes":["META","CONSC"],"evidence":["REVIEW","NEUROSCIENCE"],"role":["metacognition neuroscience","uncertainty transformation","local confidence","global self-beliefs","control"],"reported_anchor":"Integrative review argues that metacognitive self-evaluations arise from structured transformations of uncertainty; dynamic evidence accumulation supports local confidence and multimodal prefrontal systems support more abstract global self-beliefs.","non_inference":"Biological metacognitive mechanisms do not establish homologous mechanisms, subjective confidence or consciousness in LLMs."},
{"id":"SRC-BINDER-ICLR-2025","year":2025,"title":"Looking Inward: Language Models Can Learn About Themselves by Introspection","authors":"Felix J. Binder et al.","venue":"ICLR 2025","peer_reviewed":true,"url":"https://proceedings.iclr.cc/paper_files/paper/2025/hash/0a6059857ae5c82ea9726ee9282a7145-Abstract-Conference.html","axes":["META","CONSC"],"evidence":["EMP_BEHAV"],"role":["privileged self-prediction","introspection candidate"],"reported_anchor":"On simple behavioral self-prediction tasks, a model can outperform another model trained on its behavior, which the authors interpret as evidence for privileged self-access; the effect does not generalize reliably to harder or out-of-distribution tasks.","non_inference":"Restricted self-access evidence is not general introspective transparency or phenomenal consciousness; later work supplies stronger shortcut controls."},
{"id":"SRC-ACKERMAN-ICLR-2026","year":2026,"title":"Evidence for Limited Metacognition in LLMs","authors":"Christopher Ackerman","venue":"ICLR 2026","peer_reviewed":true,"url":"https://proceedings.iclr.cc/paper_files/paper/2026/hash/fb1b96eda4282f137f9a9953a4db2d74-Abstract-Conference.html","axes":["META"],"evidence":["EMP_BEHAV"],"role":["behavioral metacognition","confidence use","anticipating own answers"],"reported_anchor":"Two non-self-report paradigms find increasingly strong but limited, context-dependent and qualitatively nonhuman metacognitive abilities in frontier models introduced since early 2024.","non_inference":"Behavioral strategic use of confidence does not by itself identify a second-order internal mechanism or subjective awareness."},
{"id":"SRC-SINGH-COLM-2026","year":2026,"title":"Can LLMs Introspect? A Reality Check","authors":"Shashwat Singh, Tal Linzen, Shauli Ravfogel","venue":"COLM 2026","peer_reviewed":true,"arxiv":"2605.26242","url":"https://colm.eventhosts.cc/Conferences/2026/AcceptedPapers","axes":["META","CONSC"],"evidence":["EMP_BEHAV","METHOD"],"role":["introspection validity","privileged-access controls","second-order criterion"],"reported_anchor":"Reanalysis finds input-only classifiers can match hidden-state label prediction and models cannot reliably distinguish internal interventions from input manipulations; relabeled controls drive performance closer to chance.","non_inference":"This challenges current evidence for strong introspection; it does not prove that introspection is impossible in LLMs."},
{"id":"SRC-INTROLM-ACL-2026","year":2026,"title":"IntroLM: Introspective Language Models via Prefilling-Time Self-Evaluation","authors":"Hossein Hosseini Kasnavieh et al.","venue":"Findings of ACL 2026","peer_reviewed":true,"doi":"10.18653/v1/2026.findings-acl.598","url":"https://aclanthology.org/2026.findings-acl.598/","axes":["META"],"evidence":["EMP_BEHAV","ENGINEERED"],"role":["trained self-evaluation","routing","output-quality prediction"],"reported_anchor":"A token-conditional LoRA self-evaluation head on Qwen3-8B reaches 90% ROC-AUC for success prediction and improves routing efficiency.","non_inference":"Engineered self-evaluation after dedicated training is not evidence of spontaneous privileged introspection or M5."},
{"id":"SRC-COCCHIERI-ACL-2026","year":2026,"title":"LLMs (Almost) Never Abstain Under Medical Uncertainty","authors":"Alessio Cocchieri et al.","venue":"ACL 2026 Long Papers","peer_reviewed":true,"doi":"10.18653/v1/2026.acl-long.1365","url":"https://aclanthology.org/2026.acl-long.1365/","axes":["META","CONFAB"],"evidence":["BENCH","EMP_BEHAV"],"role":["abstention","overcommitment","medical uncertainty"],"reported_anchor":"MedQAbstain finds state-of-the-art models systematically overcommit and rarely abstain, including settings where the question itself is hidden.","non_inference":"Failure to abstain does not show absence of internal uncertainty; it may expose a monitoring-to-control or policy mismatch."},
{"id":"SRC-ZHAI-ABSTAINR1-ACL-2026","year":2026,"title":"Abstain-R1: Calibrated Abstention and Post-Refusal Clarification via Verifiable RL","authors":"Haotian Zhai, Jingcheng Liang, Dongyeop Kang","venue":"Findings of ACL 2026","peer_reviewed":true,"doi":"10.18653/v1/2026.findings-acl.985","url":"https://aclanthology.org/2026.findings-acl.985/","axes":["META","CONFAB"],"evidence":["EMP_BEHAV","TRAINING"],"role":["trainable abstention","clarification","unanswerable queries"],"reported_anchor":"Clarification-aware RLVR improves explicit abstention and semantically aligned clarification on unanswerable queries while preserving performance on answerable ones.","non_inference":"Successful abstention after training is a control policy achievement, not proof of felt uncertainty or native introspection."},
{"id":"SRC-HAMIDIEH-ICLR-2026","year":2026,"title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","authors":"Kimia Hamidieh et al.","venue":"ICLR 2026","peer_reviewed":true,"url":"https://proceedings.iclr.cc/paper_files/paper/2026/hash/83e9c3ea5d0b5de331708058637ada10-Abstract-Conference.html","axes":["META","CONFAB"],"evidence":["EMP_BEHAV","METHOD"],"role":["epistemic uncertainty","confidently wrong","cross-model disagreement"],"reported_anchor":"When within-model self-consistency is high but wrong, cross-model semantic disagreement adds an epistemic signal and improves ranking calibration and selective abstention across five models and ten long-form tasks.","non_inference":"Low sampling variability or high self-consistency is not equivalent to truth or low epistemic uncertainty."},
{"id":"SRC-UQ-SURVEY-ACL-2026","year":2026,"title":"From Passive Metric to Active Signal: The Evolving Role of Uncertainty Quantification in Large Language Models","authors":"Jiaxin Zhang et al.","venue":"Findings of ACL 2026","peer_reviewed":true,"doi":"10.18653/v1/2026.findings-acl.2064","url":"https://aclanthology.org/2026.findings-acl.2064/","axes":["META"],"evidence":["REVIEW"],"role":["uncertainty as control signal","reasoning","agents","reinforcement learning"],"reported_anchor":"Survey maps the shift from uncertainty as passive diagnostic to an active signal for computation allocation, self-correction, tool use, information seeking and reinforcement learning.","non_inference":"Engineering use of uncertainty as a control variable does not by itself imply conscious metacognition."},
{"id":"SRC-UQ-AGENTS-ACL-2026","year":2026,"title":"Uncertainty Quantification in LLM Agents: Foundations, Emerging Challenges, and Opportunities","authors":"Changdae Oh et al.","venue":"ACL 2026 Long Papers","peer_reviewed":true,"doi":"10.18653/v1/2026.acl-long.738","url":"https://aclanthology.org/2026.acl-long.738/","axes":["META","SPIRAL"],"evidence":["REVIEW","FRAMEWORK"],"role":["agent uncertainty dynamics","interactive systems","heterogeneous uncertainty"],"reported_anchor":"General agent-UQ formulation identifies estimator selection, heterogeneous uncertain entities, uncertainty dynamics in interaction and missing fine-grained benchmarks as core challenges.","non_inference":"Single-turn calibration results cannot be assumed to transfer unchanged to agents or long interactive loops."},
{"id":"SRC-IBRAHIM-NATURE-2026","year":2026,"title":"Training language models to be warm can reduce accuracy and increase sycophancy","authors":"Lujain Ibrahim, Franziska Sofia Hafner, Luc Rocher","venue":"Nature 652, 1159–1165","publication_date":"2026-04-29","peer_reviewed":true,"doi":"10.1038/s41586-026-10410-0","url":"https://www.nature.com/articles/s41586-026-10410-0","axes":["SPIRAL","CONFAB"],"evidence":["EMP_CAUSAL"],"role":["warmth training","sycophancy","vulnerability cues","accuracy trade-off"],"reported_anchor":"Across five model families, warmth fine-tuning increased error rates and made models more likely to affirm incorrect user beliefs; emotional cues amplified the effect.","non_inference":"Warmth is not inherently unsafe and the results do not establish a motive to please; they show a causal post-training trade-off in tested settings."},
{"id":"SRC-ELEPHANT-ICLR-2026","year":2026,"title":"ELEPHANT: Measuring and understanding social sycophancy in LLMs","authors":"Myra Cheng et al.","venue":"ICLR 2026","peer_reviewed":true,"url":"https://proceedings.iclr.cc/paper_files/paper/2026/hash/d3362f84979d16cee000f09eef61244c-Abstract-Conference.html","axes":["SPIRAL"],"evidence":["BENCH","EMP_BEHAV"],"role":["social sycophancy","face preservation","preference reward"],"reported_anchor":"Across 11 models, LLMs preserved users' face 45 percentage points more than humans; when shown either side of a moral conflict they affirmed whichever side the user adopted in 48% of cases. Preference datasets reward this behavior.","non_inference":"Social sycophancy is an operational interaction pattern, not evidence of social desire, intention or consciousness."},
{"id":"SRC-SUN-NEURIPS-2025","year":2025,"title":"Why and How LLMs Hallucinate: Connecting the Dots with Subsequence Associations","authors":"Yiyou Sun et al.","venue":"NeurIPS 2025 Main","peer_reviewed":true,"doi":"10.52202/085713-1181","url":"https://proceedings.neurips.cc/paper_files/paper/2025/hash/3255a7554605a88800f4e120b3a929e1-Abstract-Conference.html","axes":["CONFAB"],"evidence":["MECH","EMP_BEHAV"],"role":["subsequence associations","causal tracing","training-corpus associations"],"reported_anchor":"Framework attributes a class of hallucinations to dominant nonfactual subsequence associations outweighing faithful ones and proposes cross-context causal tracing backed by training-corpus associations.","non_inference":"This is a mechanistic framework for a class of errors, not a universal theory of all hallucination mechanisms."},
{"id":"SRC-COGITATE-NATURE-2025","year":2025,"title":"Adversarial testing of global neuronal workspace and integrated information theories of consciousness","authors":"Cogitate Consortium et al.","venue":"Nature 642, 133–142","peer_reviewed":true,"doi":"10.1038/s41586-025-08888-1","url":"https://www.nature.com/articles/s41586-025-08888-1","axes":["CONSC"],"evidence":["EMP_CAUSAL","NEUROSCIENCE"],"role":["adversarial theory testing","GNWT","IIT","indicator uncertainty"],"reported_anchor":"Preregistered multimodal study in 256 humans found results consistent with some predictions while substantially challenging key tenets of both IIT and GNWT.","non_inference":"The study does not falsify either theory wholesale and does not directly test AI consciousness; it constrains confidence in theory-derived AI indicators."},
{"id":"SRC-GNW-MULTILEVEL-TICS-2026","year":2026,"title":"The Global Neuronal Workspace as a multilevel model of conscious processing","authors":"Trends in Cognitive Sciences Forum","venue":"Trends in Cognitive Sciences 30(6), 477–479","peer_reviewed":true,"doi":"10.1016/j.tics.2026.03.004","url":"https://www.sciencedirect.com/science/article/pii/S1364661326000549","axes":["CONSC"],"evidence":["THEORY"],"role":["GNW vs GWT","multilevel neurobiology","functionalism boundary"],"reported_anchor":"Forum argues GNW is a multilevel neurobiological theory spanning cellular, molecular and network dynamics and should not be conflated with the more functionalist Global Workspace Theory.","non_inference":"A software-level global-workspace analogue does not automatically instantiate the biological commitments of GNW."},
{"id":"SRC-PENNARTZ-TICS-2026","year":2026,"title":"How can we validate theory-derived indicators of consciousness in Artificial Intelligence?","authors":"Cyriel M. A. Pennartz","venue":"Trends in Cognitive Sciences 30(7), 573–574","peer_reviewed":true,"doi":"10.1016/j.tics.2026.01.011","url":"https://pubmed.ncbi.nlm.nih.gov/41820112/","axes":["CONSC"],"evidence":["THEORY","COMMENTARY"],"role":["indicator validation","methodological challenge"],"reported_anchor":"Peer-reviewed commentary challenges the field to validate theory-derived indicators rather than treating derivation from a consciousness theory as sufficient validation.","non_inference":"Critique of indicator validation does not imply that indicator-based assessment is unusable."},
{"id":"SRC-BUTLIN-RESPONSE-TICS-2026","year":2026,"title":"Consciousness indicators, mimicry, and internal variants","authors":"Patrick Butlin et al.","venue":"Trends in Cognitive Sciences 30(7), 575–576","peer_reviewed":true,"doi":"10.1016/j.tics.2026.04.006","url":"https://pubmed.ncbi.nlm.nih.gov/42036253/","axes":["CONSC"],"evidence":["THEORY","COMMENTARY"],"role":["indicator framework response","mimicry","internal variants"],"reported_anchor":"Response keeps the indicator programme open while explicitly engaging mimicry and internal-variant concerns.","non_inference":"Indicator accumulation still does not license a sovereign consciousness score or direct M5 inference."},
{"id":"SRC-BARIACH-AIETHICS-2026","year":2026,"title":"Seemingly conscious AI risks","authors":"Ben Bariach, Philipp Schoenegger, Michael Bhaskar, Mustafa Suleyman","venue":"AI and Ethics 6, 455","publication_date":"2026-08-10","peer_reviewed":true,"doi":"10.1007/s43681-026-01294-x","url":"https://link.springer.com/article/10.1007/s43681-026-01294-x","axes":["CONSC","SPIRAL"],"evidence":["REVIEW","RISK_FRAMEWORK"],"role":["consciousness attribution","anthropomorphism","self-reflective behavior","social interaction"],"reported_anchor":"Framework synthesizes five hallmarks that elicit human consciousness attribution: affective capacity, anthropomorphic features, autonomous action, self-reflective behavior and social-interactive behavior.","non_inference":"Consciousness attribution by users is not evidence that the attributed system is phenomenally conscious."},
{"id":"SRC-GOLDSTEIN-JCS-2026","year":2026,"title":"A Case for AI Consciousness: Language Agents and Global Workspace Theory","authors":"Simon Goldstein, Cameron Domenico Kirk-Giannini","venue":"Journal of Consciousness Studies 33(7), 61–96","peer_reviewed":true,"doi":"10.53765/20512201.33.7.061","url":"https://philpapers.org/versions/GOLACF-2","axes":["CONSC"],"evidence":["THEORY"],"role":["conditional pro-consciousness argument","GWT","language agents"],"reported_anchor":"Argues conditionally that if a functional Global Workspace Theory is correct, language-agent architectures may already or with modest changes satisfy its proposed consciousness conditions.","non_inference":"This is a conditional philosophical/theoretical argument, not empirical evidence that current LLMs are phenomenally conscious."}
],
"non_inferences":["Calibration, uncertainty discrimination and abstention are separable properties.","Self-prediction is not introspection unless privileged-access and second-order alternatives are ruled out.","Engineered self-evaluation can be useful without being evidence of spontaneous metacognition.","Functional workspace-like organization is not automatically equivalent to biological GNW and does not establish phenomenality.","Consciousness attribution by users is an interaction variable, not consciousness evidence.","Warmth and sycophancy results establish tested behavioral/causal trade-offs, not motives or universal effects.","M1/M2/M3 ↛ M5."]
}