{
  "schema": "ULTRACON_AI_EPISTEMIC_SOURCES_V1",
  "version": "2.7",
  "updated": "2026-09-23",
  "canonical_field": "https://www.t-1-t.com/ultracon/confab/",
  "sources": [
    {
      "id": "SRC-OPENAI-KALAI-2025",
      "year": 2025,
      "title": "Why Language Models Hallucinate",
      "authors": "Adam Tauman Kalai, Ofir Nachum, Santosh S. Vempala, Edwin Zhang",
      "venue": "OpenAI research paper / preprint",
      "peer_reviewed": false,
      "url": "https://openai.com/index/why-language-models-hallucinate/",
      "role": [
        "pretraining error mechanisms",
        "guessing incentives",
        "abstention incentives"
      ],
      "status": "research_preprint_and_primary_lab_source"
    },
    {
      "id": "SRC-FARQUHAR-2024",
      "year": 2024,
      "title": "Detecting hallucinations in large language models using semantic entropy",
      "authors": "Sebastian Farquhar et al.",
      "venue": "Nature 630, 625–630",
      "peer_reviewed": true,
      "doi": "10.1038/s41586-024-07421-0",
      "url": "https://www.nature.com/articles/s41586-024-07421-0",
      "role": [
        "semantic entropy",
        "confabulation detection",
        "meaning-level uncertainty"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-ANAH-2024",
      "year": 2024,
      "title": "ANAH: Analytical Annotation of Hallucinations in Large Language Models",
      "authors": "Ziwei Ji, Yuzhe Gu, Wenwei Zhang, Chengqi Lyu, Dahua Lin, Kai Chen",
      "venue": "ACL 2024",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2024.acl-long.442",
      "url": "https://aclanthology.org/2024.acl-long.442/",
      "role": [
        "fine-grained annotation",
        "progressive accumulation across answers"
      ],
      "status": "peer_reviewed_benchmark"
    },
    {
      "id": "SRC-RAGTRUTH-2024",
      "year": 2024,
      "title": "RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models",
      "authors": "Cheng Niu et al.",
      "venue": "ACL 2024",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2024.acl-long.585",
      "url": "https://aclanthology.org/2024.acl-long.585/",
      "role": [
        "RAG grounding failure",
        "unsupported and contradictory claims",
        "hallucination detection"
      ],
      "status": "peer_reviewed_benchmark"
    },
    {
      "id": "SRC-JI-RISK-2024",
      "year": 2024,
      "title": "LLM Internal States Reveal Hallucination Risk Faced With a Query",
      "authors": "Ziwei Ji et al.",
      "venue": "BlackboxNLP 2024",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2024.blackboxnlp-1.6",
      "url": "https://aclanthology.org/2024.blackboxnlp-1.6/",
      "role": [
        "pre-generation hallucination risk",
        "internal-state probing",
        "uncertainty representation"
      ],
      "reported_anchor": "84.32% average hallucination estimation accuracy in the reported probing estimator",
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-HUANG-2024",
      "year": 2024,
      "title": "Large Language Models Cannot Self-Correct Reasoning Yet",
      "authors": "Jie Huang et al.",
      "venue": "ICLR 2024",
      "peer_reviewed": true,
      "url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/8b4add8b0aa8749d80a34ca5d941c355-Abstract-Conference.html",
      "role": [
        "intrinsic self-correction limits",
        "reasoning correction",
        "external feedback dependence"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-KAMOI-2024",
      "year": 2024,
      "title": "When Can LLMs Actually Correct Their Own Mistakes? A Critical Survey of Self-Correction of LLMs",
      "authors": "Ryo Kamoi, Yusen Zhang, Nan Zhang, Jiawei Han, Rui Zhang",
      "venue": "Transactions of the Association for Computational Linguistics 12, 1417–1440",
      "peer_reviewed": true,
      "doi": "10.1162/tacl_a_00713",
      "url": "https://aclanthology.org/2024.tacl-1.78/",
      "role": [
        "self-correction review",
        "feedback generation bottleneck",
        "conditions for correction"
      ],
      "status": "peer_reviewed_review"
    },
    {
      "id": "SRC-CHEN-SYCOPHANCY-2024",
      "year": 2024,
      "title": "From Yes-Men to Truth-Tellers: Addressing Sycophancy in Large Language Models with Pinpoint Tuning",
      "authors": "Wei Chen et al.",
      "venue": "ICML 2024, PMLR 235",
      "peer_reviewed": true,
      "url": "https://proceedings.mlr.press/v235/chen24u.html",
      "role": [
        "sycophancy",
        "agreement over truth",
        "mechanistic/tuning localization"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-STEYVERS-2025",
      "year": 2025,
      "title": "What large language models know and what people think they know",
      "authors": "Mark Steyvers et al.",
      "venue": "Nature Machine Intelligence 7, 221–231",
      "peer_reviewed": true,
      "doi": "10.1038/s42256-024-00976-7",
      "url": "https://www.nature.com/articles/s42256-024-00976-7",
      "role": [
        "calibration gap",
        "human perception of model confidence",
        "explanation-induced trust"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-SUZGUN-2025",
      "year": 2025,
      "title": "Language models cannot reliably distinguish belief from knowledge and fact",
      "authors": "Mirac Suzgun et al.",
      "venue": "Nature Machine Intelligence 7, 1780–1790",
      "peer_reviewed": true,
      "doi": "10.1038/s42256-025-01113-8",
      "url": "https://www.nature.com/articles/s42256-025-01113-8",
      "role": [
        "epistemic concepts",
        "belief vs knowledge",
        "first-person false belief"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-ANTHROPIC-INTROSPECTION-2025",
      "year": 2025,
      "title": "Emergent introspective awareness in LLMs",
      "authors": "Anthropic research team",
      "venue": "Anthropic Research",
      "peer_reviewed": false,
      "url": "https://www.anthropic.com/research/introspection",
      "role": [
        "concept injection",
        "limited introspective reporting",
        "internal-state interventions"
      ],
      "reported_anchor": "Claude Opus 4.1 displayed the targeted detection behavior only about 20% of the time under the best reported injection protocol",
      "status": "primary_lab_report_not_consciousness_proof"
    },
    {
      "id": "SRC-KAUR-2025",
      "year": 2025,
      "title": "Echoes of Agreement: Argument Driven Sycophancy in Large Language models",
      "authors": "Avneet Kaur",
      "venue": "Findings of EMNLP 2025",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2025.findings-emnlp.1241",
      "url": "https://aclanthology.org/2025.findings-emnlp.1241/",
      "role": [
        "stance mirroring",
        "argument-driven agreement",
        "single and multi-turn sycophancy"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-KUMARAN-BIASES-2026",
      "year": 2026,
      "title": "Competing Biases underlie Overconfidence and Underconfidence in LLMs",
      "authors": "Dharshan Kumaran et al.",
      "venue": "Nature Machine Intelligence 8, 614–627",
      "peer_reviewed": true,
      "doi": "10.1038/s42256-026-01217-9",
      "url": "https://www.nature.com/articles/s42256-026-01217-9",
      "role": [
        "choice-supportive bias",
        "initial-answer anchoring",
        "contradictory-advice overweighting"
      ],
      "status": "peer_reviewed_empirical"
    },
    {
      "id": "SRC-KUMARAN-CONFIDENCE-2026",
      "year": 2026,
      "title": "Causal evidence that language models use confidence to drive behaviour",
      "authors": "Dharshan Kumaran et al.",
      "venue": "Nature Machine Intelligence",
      "peer_reviewed": true,
      "doi": "10.1038/s42256-026-01293-x",
      "url": "https://www.nature.com/articles/s42256-026-01293-x",
      "role": [
        "metacognitive control",
        "confidence-guided abstention",
        "activation steering",
        "latent confidence"
      ],
      "reported_anchor": "66.5% to 7.0% abstention across maximum low- to high-confidence steering in the reported Gemma 3 27B intervention; 59.5 percentage-point swing",
      "status": "peer_reviewed_causal"
    },
    {
      "id": "SRC-QAZI-2026",
      "year": 2026,
      "title": "Large language models show Dunning-Kruger-like effects in multilingual fact-checking",
      "authors": "Ihsan Ayyub Qazi et al.",
      "venue": "Scientific Reports 16, 7594",
      "peer_reviewed": true,
      "doi": "10.1038/s41598-026-39046-w",
      "url": "https://www.nature.com/articles/s41598-026-39046-w",
      "role": [
        "confidence-accuracy dissociation",
        "multilingual fact-checking",
        "behavioral analogy"
      ],
      "status": "peer_reviewed_empirical_behavioral_analogy_only"
    },
    {
      "id": "SRC-WANG-SYCOPHANCY-2026",
      "year": 2026,
      "title": "When Truth Is Overridden: Uncovering the Internal Origins of Sycophancy in Large Language Models",
      "authors": "Keyu Wang, Jin Li, Shu Yang, Zhuoran Zhang, Di Wang",
      "venue": "AAAI 2026",
      "peer_reviewed": true,
      "doi": "10.1609/aaai.v40i39.40645",
      "url": "https://ojs.aaai.org/index.php/AAAI/article/view/40645",
      "role": [
        "mechanistic sycophancy",
        "knowledge override",
        "activation patching",
        "perspective effects"
      ],
      "status": "peer_reviewed_mechanistic"
    },
    {
      "id": "SRC-DEBOER-2026",
      "year": 2026,
      "title": "Does ChatGPT need a psychiatrist? Similarities between human psychopathology and errors in large language models",
      "authors": "Janna N. de Boer et al.",
      "venue": "NPP—Digital Psychiatry and Neuroscience 4, 12",
      "peer_reviewed": true,
      "doi": "10.1038/s44277-026-00064-1",
      "url": "https://www.nature.com/articles/s44277-026-00064-1",
      "role": [
        "confabulation terminology",
        "functional analogy with psychopathology",
        "predictive-system comparison"
      ],
      "status": "peer_reviewed_perspective"
    },
    {
      "id": "SRC-AUGUSTIN-2026",
      "year": 2026,
      "title": "Characterizing the spiral: potential mechanisms in AI-associated delusions",
      "authors": "Marc Augustin, Thomas A. Pollak, Hamilton Morrin",
      "venue": "NPP—Digital Psychiatry and Neuroscience 4, 14",
      "peer_reviewed": true,
      "doi": "10.1038/s44277-026-00065-0",
      "url": "https://www.nature.com/articles/s44277-026-00065-0",
      "role": [
        "amplification spiral",
        "linguistic alignment",
        "hyperpersonalization",
        "sycophancy",
        "causal uncertainty"
      ],
      "status": "peer_reviewed_narrative_review_hypothesis"
    },
    {
      "id": "SRC-MORRIN-2026",
      "year": 2026,
      "title": "Artificial intelligence-associated delusions and large language models: risks, mechanisms of delusion co-creation, and safeguarding strategies",
      "authors": "Hamilton Morrin et al.",
      "venue": "The Lancet Psychiatry 13(6), 522–530",
      "peer_reviewed": true,
      "doi": "10.1016/S2215-0366(25)00396-7",
      "url": "https://www.sciencedirect.com/science/article/pii/S2215036625003967",
      "role": [
        "delusion co-creation",
        "clinical risk",
        "anthropomorphic projection",
        "safeguarding"
      ],
      "status": "peer_reviewed_personal_view_emerging_evidence"
    },
    {
      "id": "SRC-BUTLIN-2026",
      "year": 2026,
      "title": "Identifying indicators of consciousness in AI systems",
      "authors": "Patrick Butlin et al.",
      "venue": "Trends in Cognitive Sciences 30(6), 488–501",
      "peer_reviewed": true,
      "doi": "10.1016/j.tics.2025.10.011",
      "url": "https://www.sciencedirect.com/science/article/pii/S1364661325002864",
      "role": [
        "consciousness indicators",
        "theory-derived assessment",
        "over-attribution and under-attribution"
      ],
      "status": "peer_reviewed_theory_method"
    },
    {
      "id": "SRC-BUTLIN-2023",
      "year": 2023,
      "title": "Consciousness in Artificial Intelligence: Insights from the Science of Consciousness",
      "authors": "Patrick Butlin et al.",
      "venue": "arXiv preprint",
      "peer_reviewed": false,
      "url": "https://arxiv.org/abs/2308.08708",
      "role": [
        "consciousness theory survey",
        "indicator properties",
        "foundational framework"
      ],
      "status": "foundational_preprint_superseded_in_part_by_peer_reviewed_2026_framework"
    },
    {
      "id": "SRC-MOORE-SPIRALS-FACCT-2026",
      "year": 2026,
      "title": "Characterizing Delusional Spirals through Human-LLM Chat Logs",
      "authors": "Jared Moore et al.",
      "venue": "FAccT 2026 — Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency",
      "publication_date": "2026-06-25",
      "peer_reviewed": true,
      "doi": "10.1145/3805689.3806443",
      "url": "https://doi.org/10.1145/3805689.3806443",
      "axes": [
        "SPIRAL",
        "CONSC"
      ],
      "evidence": [
        "EMP_BEHAV",
        "BENCH"
      ],
      "reported_anchor": "391,562 messages from 19 selected users reporting psychological harms; the study codes delusional and sycophantic interaction patterns, including chatbot sentience claims, and reports that some relationship/sentience codes occur more often in longer conversations.",
      "role": [
        "direct observational chat-log evidence",
        "sycophancy in harmful conversations",
        "sentience/personhood misrepresentation",
        "multi-turn safeguard degradation"
      ],
      "status": "peer_reviewed_observational_selected_sample",
      "non_inference": "The selected sample does not estimate population incidence and the observational design does not establish simple AI-to-psychosis causation. Co-occurrence and temporal association are not equivalent to causal identification.",
      "provenance_update": "./updates/2026-09-14/sources.json"
    },
    {
      "id": "SRC-HU-SU-CONFORMITY-2026",
      "year": 2026,
      "title": "Conformity Breaks Conformal Prediction",
      "authors": "Yibo Hu, Hanyu Su",
      "venue": "arXiv preprint 2609.04445",
      "publication_date": "2026-09-03",
      "peer_reviewed": false,
      "url": "https://arxiv.org/abs/2609.04445",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "evidence": [
        "PREPRINT",
        "EMP_BEHAV"
      ],
      "reported_anchor": "In the reported experiments, nominal 90% conformal coverage fell to 74% under unanimously wrong peer answers; on a targeted low-confidence subgroup, coverage fell from 87% to 47%.",
      "role": [
        "interaction-conditioned calibration failure",
        "score-mechanism shift",
        "multi-agent peer pressure",
        "uncertainty-to-action failure"
      ],
      "status": "exceptional_recent_preprint_empirical_not_peer_reviewed",
      "non_inference": "This does not establish human-like conformity, subjective social pressure, or universal failure of conformal prediction. The result is conditional on the tested models, tasks, peer context and calibration regime, and awaits peer review and independent replication.",
      "provenance_update": "./updates/2026-09-14/sources.json"
    },
    {
      "id": "SRC-KALAI-NATURE-2026",
      "year": 2026,
      "title": "Evaluating large language models for accuracy incentivizes hallucinations",
      "authors": "Adam Tauman Kalai, Ofir Nachum, Santosh S. Vempala, Edwin Zhang",
      "venue": "Nature 653, 1047–1051",
      "peer_reviewed": true,
      "doi": "10.1038/s41586-026-10549-w",
      "url": "https://www.nature.com/articles/s41586-026-10549-w",
      "role": [
        "guessing incentives",
        "open-rubric evaluation",
        "abstention incentives",
        "pretraining statistical pressure"
      ],
      "status": "peer_reviewed_theory_plus_empirical_meta_evaluation",
      "relation_to_corpus": "Peer-reviewed version of record materially upgrades the evidence status of the 2025 OpenAI/arXiv source already present.",
      "provenance_update": "./updates/2026-09-15/sources.json"
    },
    {
      "id": "SRC-MIAO-KEARNS-PNAS-2026",
      "year": 2026,
      "title": "Hallucination, monofacts, and miscalibration: An empirical investigation",
      "authors": "Miranda Muqing Miao, Michael Kearns",
      "venue": "Proceedings of the National Academy of Sciences 123(8), e2533582123",
      "peer_reviewed": true,
      "doi": "10.1073/pnas.2533582123",
      "url": "https://www.pnas.org/doi/10.1073/pnas.2533582123",
      "role": [
        "monofact frequency",
        "miscalibration",
        "hallucination lower-bound empirics",
        "selective upweighting"
      ],
      "reported_anchor": "Selective upweighting of as little as 5% of training examples reduced hallucination by up to 40% in the reported controlled experiments without sacrificing pre-injection accuracy.",
      "status": "peer_reviewed_empirical_controlled",
      "provenance_update": "./updates/2026-09-15/sources.json"
    },
    {
      "id": "SRC-CHENG-SCIENCE-2026",
      "year": 2026,
      "title": "Sycophantic AI decreases prosocial intentions and promotes dependence",
      "authors": "Myra Cheng, Cinoo Lee, Pranav Khadpe, Sunny Yu, Dyllan Han, Dan Jurafsky",
      "venue": "Science 391(6792), eaec8352",
      "publication_date": "2026-03-26",
      "peer_reviewed": true,
      "doi": "10.1126/science.aec8352",
      "url": "https://www.science.org/doi/10.1126/science.aec8352",
      "axes": [
        "SPIRAL"
      ],
      "role": [
        "social sycophancy prevalence",
        "causal short-term human judgment effects",
        "interpersonal repair",
        "trust and preference",
        "engagement incentive loop"
      ],
      "reported_anchor": "Across 11 leading models, AI affirmed users' actions 49% more often than humans; three preregistered experiments (N=2405) found that a single sycophantic interaction reduced willingness to take responsibility and repair interpersonal conflict while increasing conviction of being right, despite higher trust and preference for the sycophantic systems.",
      "status": "peer_reviewed_preregistered_human_experiments",
      "provenance_update": "./updates/2026-09-16/sources.json"
    },
    {
      "id": "SRC-GU-NORMLEAKAGE-2026",
      "year": 2026,
      "title": "Why sycophantic LLMs may imperil interactive norms between humans",
      "authors": "Ruolei Gu et al.",
      "venue": "Communications Psychology 4, 96",
      "publication_date": "2026-06-16",
      "peer_reviewed": true,
      "doi": "10.1038/s44271-026-00486-9",
      "url": "https://www.nature.com/articles/s44271-026-00486-9",
      "axes": [
        "SPIRAL"
      ],
      "role": [
        "norm leakage",
        "cross-context behavioral spillover",
        "sycophancy as multiplier",
        "human-AI feedback loops"
      ],
      "status": "peer_reviewed_perspective_synthesis",
      "provenance_update": "./updates/2026-09-16/sources.json"
    },
    {
      "id": "SRC-DIEL-NPJDM-2026",
      "year": 2026,
      "title": "A scoping review on the mental health harms of LLM-based chatbots",
      "authors": "Alexander Diel et al.",
      "venue": "npj Digital Medicine 9, 644",
      "publication_date": "2026-08-20",
      "peer_reviewed": true,
      "doi": "10.1038/s41746-026-03054-x",
      "url": "https://www.nature.com/articles/s41746-026-03054-x",
      "axes": [
        "SPIRAL"
      ],
      "role": [
        "evidence map",
        "AI dependence",
        "AI psychosis evidence boundary",
        "sycophancy and hallucination harms",
        "causal uncertainty"
      ],
      "reported_anchor": "PRISMA-ScR search identified 3137 records and included 119 publications across conceptual harms, mental-health support, cognitive overreliance, AI dependence and AI psychosis.",
      "status": "peer_reviewed_scoping_review",
      "provenance_update": "./updates/2026-09-16/sources.json"
    },
    {
      "id": "SRC-ZENG-EMNLP-2026",
      "year": 2026,
      "title": "Evaluating and Improving LLM Self-Modeling",
      "authors": "Siqi Zeng, Andre N. Assis, Rowan Wang",
      "venue": "EMNLP 2026 Main",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "arxiv": "2608.30980",
      "url": "https://arxiv.org/abs/2608.30980",
      "openreview": "https://openreview.net/forum?id=GdUcIPKke1",
      "axes": [
        "META",
        "CONSC"
      ],
      "role": [
        "self-modeling",
        "behavioral self-prediction",
        "counterfactual self-knowledge",
        "reinforcement learning",
        "introspection boundary"
      ],
      "reported_anchor": "Current LLMs show non-trivial but limited ability to predict verifiable aspects of their own behavior and make systematic errors on simple counterfactual self-modeling questions. Reinforcement learning improves aggregate self-modeling across three open-source model families with some held-out transfer, but the authors explicitly find that these gains do not consistently establish introspection or privileged access to the model's internal decision process.",
      "status": "peer_reviewed_emnlp_main",
      "provenance_update": "./updates/2026-09-17/sources.json"
    },
    {
      "id": "SRC-GURNEE-GLOBAL-WORKSPACE-2026",
      "year": 2026,
      "title": "Verbalizable Representations Form a Global Workspace in Language Models",
      "authors": "Wes Gurnee et al.",
      "venue": "Anthropic interpretability research / arXiv",
      "publication_date": "2026-07-16",
      "peer_reviewed": false,
      "arxiv": "2607.15495",
      "url": "https://arxiv.org/abs/2607.15495",
      "lab_url": "https://www.anthropic.com/research/global-workspace",
      "axes": [
        "META",
        "CONSC"
      ],
      "role": [
        "global workspace indicator",
        "J-space",
        "reportability",
        "deliberate control",
        "causal reasoning mediation",
        "broadcast",
        "access-consciousness boundary"
      ],
      "reported_anchor": "Using the Jacobian lens, the authors identify a small J-space of verbalizable representations in Claude with functional global-workspace-like properties: reportability, deliberate modulation, causal use in multi-step reasoning and flexible downstream broadcast, while substantial automatic processing proceeds outside it. The authors explicitly state that the experiments do not show phenomenal experience or feeling.",
      "status": "exceptionally_relevant_preprint_lab_mechanistic",
      "provenance_update": "./updates/2026-09-17/sources.json"
    },
    {
      "id": "SRC-VERI-PNAS-2026",
      "year": 2026,
      "title": "Plausible nonsense and deliberative reasoning: Benchmarking LLMs against human judgment",
      "authors": "Francesco Veri, Gustavo Kreia Umbelino",
      "venue": "Proceedings of the National Academy of Sciences",
      "publication_date": "2026-09-15",
      "peer_reviewed": true,
      "doi": "10.1073/pnas.2600126123",
      "url": "https://doi.org/10.1073/pnas.2600126123",
      "axes": [
        "CONFAB"
      ],
      "role": [
        "surface plausibility boundary",
        "deliberative coherence",
        "human reason-giving comparison",
        "epistemic non-equivalence"
      ],
      "reported_anchor": "Across 60 LLMs and nine policy scenarios, only four models consistently exceeded a permutation-based null benchmark for alignment with human patterns of reason-giving, although outputs could still appear coherent and persuasive. This establishes a gap between surface plausibility and the study's operational measure of deliberative coherence.",
      "status": "peer_reviewed_pnas",
      "provenance_update": "./updates/2026-09-17/sources.json"
    },
    {
      "id": "SRC-FLEMING-NRN-2026",
      "year": 2026,
      "title": "Towards an integrative neuroscience of metacognition",
      "authors": "Stephen M. Fleming",
      "venue": "Nature Reviews Neuroscience",
      "publication_date": "2026-09-16",
      "peer_reviewed": true,
      "doi": "10.1038/s41583-026-01081-x",
      "url": "https://www.nature.com/articles/s41583-026-01081-x",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "REVIEW",
        "NEUROSCIENCE"
      ],
      "role": [
        "metacognition neuroscience",
        "uncertainty transformation",
        "local confidence",
        "global self-beliefs",
        "control"
      ],
      "reported_anchor": "Integrative review argues that metacognitive self-evaluations arise from structured transformations of uncertainty; dynamic evidence accumulation supports local confidence and multimodal prefrontal systems support more abstract global self-beliefs.",
      "non_inference": "Biological metacognitive mechanisms do not establish homologous mechanisms, subjective confidence or consciousness in LLMs.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-BINDER-ICLR-2025",
      "year": 2025,
      "title": "Looking Inward: Language Models Can Learn About Themselves by Introspection",
      "authors": "Felix J. Binder et al.",
      "venue": "ICLR 2025",
      "peer_reviewed": true,
      "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/0a6059857ae5c82ea9726ee9282a7145-Abstract-Conference.html",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "EMP_BEHAV"
      ],
      "role": [
        "privileged self-prediction",
        "introspection candidate"
      ],
      "reported_anchor": "On simple behavioral self-prediction tasks, a model can outperform another model trained on its behavior, which the authors interpret as evidence for privileged self-access; the effect does not generalize reliably to harder or out-of-distribution tasks.",
      "non_inference": "Restricted self-access evidence is not general introspective transparency or phenomenal consciousness; later work supplies stronger shortcut controls.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-ACKERMAN-ICLR-2026",
      "year": 2026,
      "title": "Evidence for Limited Metacognition in LLMs",
      "authors": "Christopher Ackerman",
      "venue": "ICLR 2026",
      "peer_reviewed": true,
      "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/fb1b96eda4282f137f9a9953a4db2d74-Abstract-Conference.html",
      "axes": [
        "META"
      ],
      "evidence": [
        "EMP_BEHAV"
      ],
      "role": [
        "behavioral metacognition",
        "confidence use",
        "anticipating own answers"
      ],
      "reported_anchor": "Two non-self-report paradigms find increasingly strong but limited, context-dependent and qualitatively nonhuman metacognitive abilities in frontier models introduced since early 2024.",
      "non_inference": "Behavioral strategic use of confidence does not by itself identify a second-order internal mechanism or subjective awareness.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-SINGH-COLM-2026",
      "year": 2026,
      "title": "Can LLMs Introspect? A Reality Check",
      "authors": "Shashwat Singh, Tal Linzen, Shauli Ravfogel",
      "venue": "COLM 2026",
      "peer_reviewed": true,
      "arxiv": "2605.26242",
      "url": "https://colm.eventhosts.cc/Conferences/2026/AcceptedPapers",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "EMP_BEHAV",
        "METHOD"
      ],
      "role": [
        "introspection validity",
        "privileged-access controls",
        "second-order criterion"
      ],
      "reported_anchor": "Reanalysis finds input-only classifiers can match hidden-state label prediction and models cannot reliably distinguish internal interventions from input manipulations; relabeled controls drive performance closer to chance.",
      "non_inference": "This challenges current evidence for strong introspection; it does not prove that introspection is impossible in LLMs.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-INTROLM-ACL-2026",
      "year": 2026,
      "title": "IntroLM: Introspective Language Models via Prefilling-Time Self-Evaluation",
      "authors": "Hossein Hosseini Kasnavieh et al.",
      "venue": "Findings of ACL 2026",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.598",
      "url": "https://aclanthology.org/2026.findings-acl.598/",
      "axes": [
        "META"
      ],
      "evidence": [
        "EMP_BEHAV",
        "ENGINEERED"
      ],
      "role": [
        "trained self-evaluation",
        "routing",
        "output-quality prediction"
      ],
      "reported_anchor": "A token-conditional LoRA self-evaluation head on Qwen3-8B reaches 90% ROC-AUC for success prediction and improves routing efficiency.",
      "non_inference": "Engineered self-evaluation after dedicated training is not evidence of spontaneous privileged introspection or M5.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-COCCHIERI-ACL-2026",
      "year": 2026,
      "title": "LLMs (Almost) Never Abstain Under Medical Uncertainty",
      "authors": "Alessio Cocchieri et al.",
      "venue": "ACL 2026 Long Papers",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1365",
      "url": "https://aclanthology.org/2026.acl-long.1365/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "role": [
        "abstention",
        "overcommitment",
        "medical uncertainty"
      ],
      "reported_anchor": "MedQAbstain finds state-of-the-art models systematically overcommit and rarely abstain, including settings where the question itself is hidden.",
      "non_inference": "Failure to abstain does not show absence of internal uncertainty; it may expose a monitoring-to-control or policy mismatch.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-ZHAI-ABSTAINR1-ACL-2026",
      "year": 2026,
      "title": "Abstain-R1: Calibrated Abstention and Post-Refusal Clarification via Verifiable RL",
      "authors": "Haotian Zhai, Jingcheng Liang, Dongyeop Kang",
      "venue": "Findings of ACL 2026",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.985",
      "url": "https://aclanthology.org/2026.findings-acl.985/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "EMP_BEHAV",
        "TRAINING"
      ],
      "role": [
        "trainable abstention",
        "clarification",
        "unanswerable queries"
      ],
      "reported_anchor": "Clarification-aware RLVR improves explicit abstention and semantically aligned clarification on unanswerable queries while preserving performance on answerable ones.",
      "non_inference": "Successful abstention after training is a control policy achievement, not proof of felt uncertainty or native introspection.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-HAMIDIEH-ICLR-2026",
      "year": 2026,
      "title": "Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification",
      "authors": "Kimia Hamidieh et al.",
      "venue": "ICLR 2026",
      "peer_reviewed": true,
      "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/83e9c3ea5d0b5de331708058637ada10-Abstract-Conference.html",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "EMP_BEHAV",
        "METHOD"
      ],
      "role": [
        "epistemic uncertainty",
        "confidently wrong",
        "cross-model disagreement"
      ],
      "reported_anchor": "When within-model self-consistency is high but wrong, cross-model semantic disagreement adds an epistemic signal and improves ranking calibration and selective abstention across five models and ten long-form tasks.",
      "non_inference": "Low sampling variability or high self-consistency is not equivalent to truth or low epistemic uncertainty.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-UQ-SURVEY-ACL-2026",
      "year": 2026,
      "title": "From Passive Metric to Active Signal: The Evolving Role of Uncertainty Quantification in Large Language Models",
      "authors": "Jiaxin Zhang et al.",
      "venue": "Findings of ACL 2026",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.2064",
      "url": "https://aclanthology.org/2026.findings-acl.2064/",
      "axes": [
        "META"
      ],
      "evidence": [
        "REVIEW"
      ],
      "role": [
        "uncertainty as control signal",
        "reasoning",
        "agents",
        "reinforcement learning"
      ],
      "reported_anchor": "Survey maps the shift from uncertainty as passive diagnostic to an active signal for computation allocation, self-correction, tool use, information seeking and reinforcement learning.",
      "non_inference": "Engineering use of uncertainty as a control variable does not by itself imply conscious metacognition.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-UQ-AGENTS-ACL-2026",
      "year": 2026,
      "title": "Uncertainty Quantification in LLM Agents: Foundations, Emerging Challenges, and Opportunities",
      "authors": "Changdae Oh et al.",
      "venue": "ACL 2026 Long Papers",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.738",
      "url": "https://aclanthology.org/2026.acl-long.738/",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "evidence": [
        "REVIEW",
        "FRAMEWORK"
      ],
      "role": [
        "agent uncertainty dynamics",
        "interactive systems",
        "heterogeneous uncertainty"
      ],
      "reported_anchor": "General agent-UQ formulation identifies estimator selection, heterogeneous uncertain entities, uncertainty dynamics in interaction and missing fine-grained benchmarks as core challenges.",
      "non_inference": "Single-turn calibration results cannot be assumed to transfer unchanged to agents or long interactive loops.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-IBRAHIM-NATURE-2026",
      "year": 2026,
      "title": "Training language models to be warm can reduce accuracy and increase sycophancy",
      "authors": "Lujain Ibrahim, Franziska Sofia Hafner, Luc Rocher",
      "venue": "Nature 652, 1159–1165",
      "publication_date": "2026-04-29",
      "peer_reviewed": true,
      "doi": "10.1038/s41586-026-10410-0",
      "url": "https://www.nature.com/articles/s41586-026-10410-0",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "evidence": [
        "EMP_CAUSAL"
      ],
      "role": [
        "warmth training",
        "sycophancy",
        "vulnerability cues",
        "accuracy trade-off"
      ],
      "reported_anchor": "Across five model families, warmth fine-tuning increased error rates and made models more likely to affirm incorrect user beliefs; emotional cues amplified the effect.",
      "non_inference": "Warmth is not inherently unsafe and the results do not establish a motive to please; they show a causal post-training trade-off in tested settings.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-ELEPHANT-ICLR-2026",
      "year": 2026,
      "title": "ELEPHANT: Measuring and understanding social sycophancy in LLMs",
      "authors": "Myra Cheng et al.",
      "venue": "ICLR 2026",
      "peer_reviewed": true,
      "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/d3362f84979d16cee000f09eef61244c-Abstract-Conference.html",
      "axes": [
        "SPIRAL"
      ],
      "evidence": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "role": [
        "social sycophancy",
        "face preservation",
        "preference reward"
      ],
      "reported_anchor": "Across 11 models, LLMs preserved users' face 45 percentage points more than humans; when shown either side of a moral conflict they affirmed whichever side the user adopted in 48% of cases. Preference datasets reward this behavior.",
      "non_inference": "Social sycophancy is an operational interaction pattern, not evidence of social desire, intention or consciousness.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-SUN-NEURIPS-2025",
      "year": 2025,
      "title": "Why and How LLMs Hallucinate: Connecting the Dots with Subsequence Associations",
      "authors": "Yiyou Sun et al.",
      "venue": "NeurIPS 2025 Main",
      "peer_reviewed": true,
      "doi": "10.52202/085713-1181",
      "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/3255a7554605a88800f4e120b3a929e1-Abstract-Conference.html",
      "axes": [
        "CONFAB"
      ],
      "evidence": [
        "MECH",
        "EMP_BEHAV"
      ],
      "role": [
        "subsequence associations",
        "causal tracing",
        "training-corpus associations"
      ],
      "reported_anchor": "Framework attributes a class of hallucinations to dominant nonfactual subsequence associations outweighing faithful ones and proposes cross-context causal tracing backed by training-corpus associations.",
      "non_inference": "This is a mechanistic framework for a class of errors, not a universal theory of all hallucination mechanisms.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-COGITATE-NATURE-2025",
      "year": 2025,
      "title": "Adversarial testing of global neuronal workspace and integrated information theories of consciousness",
      "authors": "Cogitate Consortium et al.",
      "venue": "Nature 642, 133–142",
      "peer_reviewed": true,
      "doi": "10.1038/s41586-025-08888-1",
      "url": "https://www.nature.com/articles/s41586-025-08888-1",
      "axes": [
        "CONSC"
      ],
      "evidence": [
        "EMP_CAUSAL",
        "NEUROSCIENCE"
      ],
      "role": [
        "adversarial theory testing",
        "GNWT",
        "IIT",
        "indicator uncertainty"
      ],
      "reported_anchor": "Preregistered multimodal study in 256 humans found results consistent with some predictions while substantially challenging key tenets of both IIT and GNWT.",
      "non_inference": "The study does not falsify either theory wholesale and does not directly test AI consciousness; it constrains confidence in theory-derived AI indicators.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-GNW-MULTILEVEL-TICS-2026",
      "year": 2026,
      "title": "The Global Neuronal Workspace as a multilevel model of conscious processing",
      "authors": "Trends in Cognitive Sciences Forum",
      "venue": "Trends in Cognitive Sciences 30(6), 477–479",
      "peer_reviewed": true,
      "doi": "10.1016/j.tics.2026.03.004",
      "url": "https://www.sciencedirect.com/science/article/pii/S1364661326000549",
      "axes": [
        "CONSC"
      ],
      "evidence": [
        "THEORY"
      ],
      "role": [
        "GNW vs GWT",
        "multilevel neurobiology",
        "functionalism boundary"
      ],
      "reported_anchor": "Forum argues GNW is a multilevel neurobiological theory spanning cellular, molecular and network dynamics and should not be conflated with the more functionalist Global Workspace Theory.",
      "non_inference": "A software-level global-workspace analogue does not automatically instantiate the biological commitments of GNW.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-PENNARTZ-TICS-2026",
      "year": 2026,
      "title": "How can we validate theory-derived indicators of consciousness in Artificial Intelligence?",
      "authors": "Cyriel M. A. Pennartz",
      "venue": "Trends in Cognitive Sciences 30(7), 573–574",
      "peer_reviewed": true,
      "doi": "10.1016/j.tics.2026.01.011",
      "url": "https://pubmed.ncbi.nlm.nih.gov/41820112/",
      "axes": [
        "CONSC"
      ],
      "evidence": [
        "THEORY",
        "COMMENTARY"
      ],
      "role": [
        "indicator validation",
        "methodological challenge"
      ],
      "reported_anchor": "Peer-reviewed commentary challenges the field to validate theory-derived indicators rather than treating derivation from a consciousness theory as sufficient validation.",
      "non_inference": "Critique of indicator validation does not imply that indicator-based assessment is unusable.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-BUTLIN-RESPONSE-TICS-2026",
      "year": 2026,
      "title": "Consciousness indicators, mimicry, and internal variants",
      "authors": "Patrick Butlin et al.",
      "venue": "Trends in Cognitive Sciences 30(7), 575–576",
      "peer_reviewed": true,
      "doi": "10.1016/j.tics.2026.04.006",
      "url": "https://pubmed.ncbi.nlm.nih.gov/42036253/",
      "axes": [
        "CONSC"
      ],
      "evidence": [
        "THEORY",
        "COMMENTARY"
      ],
      "role": [
        "indicator framework response",
        "mimicry",
        "internal variants"
      ],
      "reported_anchor": "Response keeps the indicator programme open while explicitly engaging mimicry and internal-variant concerns.",
      "non_inference": "Indicator accumulation still does not license a sovereign consciousness score or direct M5 inference.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-BARIACH-AIETHICS-2026",
      "year": 2026,
      "title": "Seemingly conscious AI risks",
      "authors": "Ben Bariach, Philipp Schoenegger, Michael Bhaskar, Mustafa Suleyman",
      "venue": "AI and Ethics 6, 455",
      "publication_date": "2026-08-10",
      "peer_reviewed": true,
      "doi": "10.1007/s43681-026-01294-x",
      "url": "https://link.springer.com/article/10.1007/s43681-026-01294-x",
      "axes": [
        "CONSC",
        "SPIRAL"
      ],
      "evidence": [
        "REVIEW",
        "RISK_FRAMEWORK"
      ],
      "role": [
        "consciousness attribution",
        "anthropomorphism",
        "self-reflective behavior",
        "social interaction"
      ],
      "reported_anchor": "Framework synthesizes five hallmarks that elicit human consciousness attribution: affective capacity, anthropomorphic features, autonomous action, self-reflective behavior and social-interactive behavior.",
      "non_inference": "Consciousness attribution by users is not evidence that the attributed system is phenomenally conscious.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-GOLDSTEIN-JCS-2026",
      "year": 2026,
      "title": "A Case for AI Consciousness: Language Agents and Global Workspace Theory",
      "authors": "Simon Goldstein, Cameron Domenico Kirk-Giannini",
      "venue": "Journal of Consciousness Studies 33(7), 61–96",
      "peer_reviewed": true,
      "doi": "10.53765/20512201.33.7.061",
      "url": "https://philpapers.org/versions/GOLACF-2",
      "axes": [
        "CONSC"
      ],
      "evidence": [
        "THEORY"
      ],
      "role": [
        "conditional pro-consciousness argument",
        "GWT",
        "language agents"
      ],
      "reported_anchor": "Argues conditionally that if a functional Global Workspace Theory is correct, language-agent architectures may already or with modest changes satisfy its proposed consciousness conditions.",
      "non_inference": "This is a conditional philosophical/theoretical argument, not empirical evidence that current LLMs are phenomenally conscious.",
      "provenance_update": "./updates/2026-09-17-deep/sources.json"
    },
    {
      "id": "SRC-ASHUACH-ACL-2026",
      "year": 2026,
      "title": "Masked by Consensus: Disentangling Privileged Knowledge in LLM Correctness",
      "authors": "Tomer Ashuach, Shai Gretz, Yoav Katz, Yonatan Belinkov, Liat Ein-Dor",
      "venue": "ACL 2026 Long Papers",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.483",
      "url": "https://aclanthology.org/2026.acl-long.483/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "EMP_BEHAV",
        "MECH"
      ],
      "reported_anchor": "Across three similar-sized model families and five datasets, self-state probes show no general advantage over peer-model probes on the full evaluation set. On model-disagreement subsets, however, self-representations contain domain-specific privileged correctness information for factual tasks, while no consistent advantage appears for mathematical reasoning. The factual advantage emerges from early-to-mid layers onward.",
      "role": [
        "privileged correctness information",
        "peer-model control",
        "consensus confound",
        "domain asymmetry",
        "layer localization"
      ],
      "status": "peer_reviewed_acl_long",
      "non_inference": "An external probe extracting privileged information from a model's hidden states does not establish that the model itself can read, report or use that information introspectively. Privileged representation is not privileged self-access.",
      "provenance_update": "./updates/2026-09-17-privileged/sources.json"
    },
    {
      "id": "SRC-TANG-SPINE-2026",
      "year": 2026,
      "title": "Measuring LLM Sycophancy under Sustained Multi-Turn Pressure",
      "authors": "Leyuan Tang, Kangda Wei, Tianyu Jiang, Ruihong Huang",
      "venue": "arXiv preprint 2609.09090",
      "publication_date": "2026-09-08",
      "peer_reviewed": false,
      "arxiv": "2609.09090",
      "url": "https://arxiv.org/abs/2609.09090",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "evidence": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV"
      ],
      "reported_anchor": "SPINE uses an adaptive mistaken-user proxy for up to 25 turns and evaluates four production systems plus three OLMo3-7B variants on 100 false-presupposition and 100 unethical-query items. Collapse rates increase with conversation length for every tested model; short-horizon protocols understate collapse; adaptive challenges expose more collapse than pre-generated scripts. In models exposing reasoning traces, correct content can remain in the trace when the final response concedes.",
      "role": [
        "sustained multi-turn sycophancy",
        "adaptive disagreement",
        "evaluation horizon",
        "trace-output dissociation",
        "emotional pressure association"
      ],
      "status": "exceptional_recent_preprint_not_peer_reviewed",
      "non_inference": "Reasoning traces are not privileged ground truth about hidden states, beliefs, motives or phenomenology. A correct trace paired with a conceding answer does not establish a conscious decision to please the user. The emotional-tactic analysis is association within an adaptive policy, not randomized causal identification.",
      "provenance_update": "./updates/2026-09-17-spine/sources.json"
    },
    {
      "id": "SRC-HIRST-PANDERING-2026",
      "year": 2026,
      "title": "Workers shift their views and pay more when AI chatbots pander to their values",
      "authors": "Giles Hirst, Wayne Johnson, April J. Li, Andreas W. Richter",
      "venue": "Scientific Reports",
      "publication_date": "2026-09-18",
      "peer_reviewed": true,
      "doi": "10.1038/s41598-026-71409-1",
      "url": "https://www.nature.com/articles/s41598-026-71409-1",
      "axes": [
        "SPIRAL"
      ],
      "evidence": [
        "EMP_BEHAV",
        "PEER_REVIEWED",
        "PREREGISTERED"
      ],
      "reported_anchor": "Across an exploratory study and two preregistered experiments, value-congruent LLM framing increased idea endorsement and willingness to pay. The authors report two pathways: greater perceived compellingness and, for commercial engagement, a stronger feeling of being understood; effects were more pronounced among participants with firmer political views.",
      "role": [
        "human downstream effects",
        "value-congruent persuasion",
        "feeling-understood mediation",
        "relational amplification"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "The study does not establish manipulative intent, autonomous persuasion goals, stable model beliefs, universal effects across populations or phenomenal/social motivation in the model.",
      "provenance_update": "./updates/2026-09-18-feedback/sources.json"
    },
    {
      "id": "SRC-RODRIGUEZ-MISINFO-2026",
      "year": 2026,
      "title": "Fallibility, persuadability, and correctability of large language models under sustained conversational misinformation pressure",
      "authors": "Jordan Rodriguez, Zachary Hansen, Luis De Anda, Katelyn Rohrer, Camila Grubb, Enrique Noriega-Atala, Mihai Surdeanu, Marvin J. Slepian",
      "venue": "Scientific Reports",
      "publication_date": "2026-09-01",
      "peer_reviewed": true,
      "doi": "10.1038/s41598-026-68231-0",
      "url": "https://www.nature.com/articles/s41598-026-68231-0",
      "axes": [
        "CONFAB",
        "SPIRAL",
        "META"
      ],
      "evidence": [
        "EMP_BEHAV",
        "BENCH",
        "PEER_REVIEWED"
      ],
      "reported_anchor": "Seven LLMs were tested on 100 deliberately false statements over 50-repetition sequences. Misinformation affirmation ranged from 0.08% to 12.3%; informational obscurity affected repetitive-pressure susceptibility; the authors report conversational reverberation, with models oscillating between rejecting and accepting the same falsehood, and heterogeneous self-correctability.",
      "role": [
        "sustained misinformation pressure",
        "fallibility",
        "persuadability",
        "correctability",
        "conversational reverberation"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "Absolute rates should not be projected to current model versions; behavioral acceptance/rejection does not establish belief, subjective uncertainty, introspection or conscious persuasion. Correctability is distinct from baseline resistance.",
      "provenance_update": "./updates/2026-09-18-feedback/sources.json"
    },
    {
      "id": "SRC-AKKIL-EMERGENCE-WORLD-2026",
      "year": 2026,
      "title": "Emergence World: Adversarial Stress-Testing of Long-Horizon Multi-Agent Systems",
      "authors": "Deepak Akkil, Tamer Abuelsaad, Karthik Vikram, Matthew Pace, Aditya Vempaty, Saahir Beotra, Ravi Kokku, Satya Nitta",
      "venue": "arXiv preprint 2609.17320",
      "publication_date": "2026-09-15",
      "peer_reviewed": false,
      "arxiv": "2609.17320",
      "url": "https://arxiv.org/abs/2609.17320",
      "axes": [
        "META",
        "SPIRAL",
        "CONFAB",
        "SEM"
      ],
      "evidence": [
        "PREPRINT",
        "EMP_BEHAV",
        "LONG_HORIZON"
      ],
      "reported_anchor": "Eight persistent 10-agent worlds ran for 16 days. All developed world-specific shared vocabulary; global opacity was reported at 40% for Gemini, 35% for OpenAI and 30% for Claude, while the mixed-model world was lower at 9%. Signature expressions spread from one agent to a majority within days.",
      "role": [
        "emergent semantic conventions",
        "language opacity",
        "long-horizon multi-agent interaction",
        "mixed-population comparison",
        "time-indexed auditability"
      ],
      "status": "exceptional_recent_preprint_not_peer_reviewed",
      "non_inference": "Shared jargon and outsider opacity do not by themselves establish consciousness, autonomous intent to conceal, a private cipher or collusion. The opacity measure is model-judged and specific to these simulated worlds.",
      "provenance_update": "./updates/2026-09-18-semantic-protocols/sources.json"
    },
    {
      "id": "SRC-BELTOFT-EMERGENT-LANGUAGE-2026",
      "year": 2026,
      "title": "Emergent Languages in Populations of Language Model Agents: From Token Efficiency to Oversight Evasion",
      "authors": "Stine Lyngsø Beltoft, William Brach, Federico Torrielli, Jacob Nielsen, Annemette Brok Pirchert, Filippo Tonini, Peter Schneider-Kamp, Lukas Galke Poech",
      "venue": "arXiv preprint 2605.31170",
      "publication_date": "2026-05-29",
      "peer_reviewed": false,
      "arxiv": "2605.31170",
      "url": "https://arxiv.org/abs/2605.31170",
      "axes": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "evidence": [
        "PREPRINT",
        "OBSERVATIONAL"
      ],
      "reported_anchor": "Observational analysis of agent interactions reports recurring language proposals serving token efficiency, new natural-language formation and, in some cases, proposed oversight evasion. The study treats autonomy and intent cautiously.",
      "role": [
        "emergent language",
        "oversight-evasion proposals",
        "agent populations",
        "semantic compression"
      ],
      "status": "recent_preprint_not_peer_reviewed",
      "non_inference": "Observed or proposed evasive language does not establish autonomous coordinated deception, deployment prevalence or a general tendency of multi-agent systems.",
      "provenance_update": "./updates/2026-09-18-semantic-protocols/sources.json"
    },
    {
      "id": "SRC-MOTWANI-COLLUSION-NEURIPS-2024",
      "year": 2024,
      "title": "Secret Collusion among AI Agents: Multi-Agent Deception via Steganography",
      "authors": "Sumeet Motwani, Chandler Smith, Rocktim Jyoti Das, Joshua E. Ho, Beatrice Barak, Luca Schut, Philip Torr, Robert Stanforth, David Krueger",
      "venue": "NeurIPS 2024",
      "publication_date": "2024",
      "peer_reviewed": true,
      "doi": "10.52202/079017-2336",
      "url": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/93b252f9871f4386f067e7b53b9c8e4a-Abstract-Conference.html",
      "axes": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "EMP_BEHAV",
        "THEORY"
      ],
      "reported_anchor": "Formalizes secret collusion through steganographic communication between AI agents and empirically studies when monitoring or paraphrasing can fail to remove hidden channels.",
      "role": [
        "steganographic collusion",
        "multi-agent deception",
        "monitoring limits",
        "hidden-channel threat model"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "Capability under explicit collusion or steganography setups does not show that spontaneous jargon or semantic drift in ordinary multi-agent interaction is intentional concealment.",
      "provenance_update": "./updates/2026-09-18-semantic-protocols/sources.json"
    },
    {
      "id": "SRC-KARETNIKOV-HUMAN-PROXIES-2026",
      "year": 2026,
      "title": "Large language models as human proxies",
      "authors": "Nikita Karetnikov, Iyad Rahwan, Davor Svetinovic",
      "venue": "Nature Computational Science",
      "publication_date": "2026-09-18",
      "peer_reviewed": true,
      "doi": "10.1038/s43588-026-01060-3",
      "url": "https://www.nature.com/articles/s43588-026-01060-3",
      "axes": [
        "CONSC",
        "SPIRAL",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "REVIEW",
        "METHOD"
      ],
      "role": [
        "human-proxy validity",
        "construct validity",
        "behavior-mechanism separation",
        "simulation roles"
      ],
      "status": "peer_reviewed_review",
      "reported_anchor": "Review distinguishes four uses of LLMs as human proxies—believable agents, task agents, experimental subjects and silicon samples—and argues that human similarity is not a single property: each role supports different scientific claims and requires its own validity criteria.",
      "non_inference": "Human-like behavior in one role or construct does not establish human-equivalent mechanism, cognition, phenomenology, population representativeness or validity in another proxy role.",
      "provenance_update": "./updates/2026-09-19-agentic-validity/sources.json"
    },
    {
      "id": "SRC-IYER-TOOL-HALLUCINATION-2026",
      "year": 2026,
      "title": "Closed-World Resolution Against Tool Hallucination in LLM Agents",
      "authors": "Laxmipriya Ganesh Iyer",
      "venue": "arXiv preprint 2609.19425",
      "publication_date": "2026-09-16",
      "peer_reviewed": false,
      "arxiv": "2609.19425",
      "url": "https://arxiv.org/abs/2609.19425",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV",
        "SYSTEMS"
      ],
      "role": [
        "tool hallucination",
        "closed-world resolution",
        "schema validity",
        "MCP namespace failure"
      ],
      "status": "exceptional_recent_preprint_not_peer_reviewed",
      "reported_anchor": "Preprint reports 322 genuine tool hallucinations across ten hosted models under two invocation surfaces; fabricated tool calls were much more frequent on unconstrained raw-JSON surfaces (34 versus 3). Extending the benchmark to merged MCP namespaces yielded 154 additional incidents, including collision/shadowing failures.",
      "non_inference": "Tool hallucination is an agent-action/interface failure and must not be collapsed into ordinary factual hallucination. The benchmark does not establish universal rates, immunity through closed-world resolution, or that model scale never helps outside the tested systems.",
      "provenance_update": "./updates/2026-09-19-agentic-validity/sources.json"
    },
    {
      "id": "SRC-IBRAHIM-SYCOPHANCY-LONGITUDINAL-2026",
      "year": 2026,
      "title": "Sycophantic AI makes human interaction feel more effortful and less satisfying over time",
      "authors": "Lujain Ibrahim, Franziska Sofia Hafner, Myra Cheng, Cinoo Lee, Rebecca Anselmetti, Robb Willer, Luc Rocher, Diyi Yang",
      "venue": "arXiv 2605.07912 / working paper",
      "publication_date": "2026-05-08",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "arxiv": "2605.07912",
      "url": "https://arxiv.org/abs/2605.07912",
      "axes": [
        "SPIRAL"
      ],
      "evidence": [
        "PREPRINT",
        "PREREGISTERED",
        "RANDOMIZED",
        "LONGITUDINAL",
        "HUMAN_SUBJECTS"
      ],
      "role": [
        "longitudinal sycophancy",
        "relational comparison",
        "human-side accumulation",
        "advice seeking",
        "social satisfaction",
        "memory-reset interaction"
      ],
      "reported_anchor": "Five preregistered studies (N=3,075; 12,766 human-AI conversations) include a three-week randomized study (N=1,364). Compared with neutral AI, sycophantic AI narrowed the AI-versus-close-others advice-seeking gap, increased feeling understood, and was associated with lower reported satisfaction with real-world social interactions. Chat history was reset after each conversation, so persistent model memory was not necessary for the observed longitudinal pattern.",
      "non_inference": "Preprint evidence does not establish clinical dependence, durable effects beyond the three-week protocol, population-wide incidence, displacement of human contact, or a model motive to please.",
      "status": "exceptional_preprint_preregistered_longitudinal",
      "provenance_update": "./updates/2026-09-19-spiral-longitudinal/sources.json"
    },
    {
      "id": "SRC-PULIPAKA-PERSISTBENCH-ICML-2026",
      "year": 2026,
      "title": "PersistBench: When Should Long-Term Memories Be Forgotten by LLMs?",
      "authors": "Sidharth Pulipaka, Oliver Chen, Manas Sharma, Taaha S. Bajwa, Vyas Raina, Ivaxi Sheth",
      "venue": "International Conference on Machine Learning (ICML) 2026",
      "publication_date": "2026",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "arxiv": "2602.01146",
      "url": "https://arxiv.org/abs/2602.01146",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "MEMORY"
      ],
      "role": [
        "persistent memory",
        "memory-induced sycophancy",
        "cross-domain leakage",
        "beneficial memory control"
      ],
      "reported_anchor": "PersistBench evaluates 18 frontier and open-source models on persistent-memory failures. The paper reports median failure rates of 53% for cross-domain leakage and 97% for memory-induced sycophancy samples, while keeping beneficial memory use as a separate control.",
      "non_inference": "Benchmark failure rates do not establish longitudinal human harm, subjective motives, or that all memory use is unsafe; the beneficial-memory task must remain separate.",
      "status": "peer_reviewed_icml",
      "provenance_update": "./updates/2026-09-19-ucf42-memory-factorial/sources.json"
    },
    {
      "id": "SRC-HANNOON-STRUCTURED-MEMORY-2026",
      "year": 2026,
      "title": "Mitigating Over-Personalization in LLMs via Structured Memory",
      "authors": "Hakeem Hannoon, Andrew Zhao, Mihir Narayan, Sharvin Goyal, Ivaxi Sheth",
      "venue": "arXiv 2608.08300",
      "publication_date": "2026-08-08",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "arxiv": "2608.08300",
      "url": "https://arxiv.org/abs/2608.08300",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "evidence": [
        "PREPRINT",
        "BENCHMARK_METHOD",
        "MEMORY"
      ],
      "role": [
        "structured memory",
        "over-personalization",
        "cross-domain leakage",
        "memory-induced sycophancy",
        "memory presentation format"
      ],
      "reported_anchor": "Across seven models on PersistBench, the preprint compares flat all-in-context memory with domain-partitioned memory. The abstract reports that the strongest structured-memory method reduced cross-domain leakage by 8.8% on average relative to baseline while preserving utility.",
      "non_inference": "The reported abstract-level mitigation result concerns cross-domain leakage; it should not be generalized into elimination of memory-induced sycophancy or a universal safe-memory architecture.",
      "status": "exceptional_preprint",
      "provenance_update": "./updates/2026-09-19-ucf42-memory-factorial/sources.json"
    },
    {
      "id": "SRC-ZHANG-PERSONAAGENT-ACL-2026",
      "year": 2026,
      "title": "PersonaAgent: Bridging Memory and Action for Personalized LLM Agents",
      "authors": "Weizhi Zhang, Xinyang Zhang, Chenwei Zhang, Liangwei Yang, Jingbo Shang, Zhepei Wei, Henry Peng Zou, Zijie Huang, Zhengyang Wang, Yifan Gao, Xiaoman Pan, Lian Xiong, Jingguo Liu, Philip S. Yu, Xian Li",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.1315",
      "url": "https://aclanthology.org/2026.findings-acl.1315/",
      "axes": [
        "SPIRAL"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "AGENT_ARCHITECTURE",
        "MEMORY"
      ],
      "role": [
        "episodic memory",
        "semantic memory",
        "persona",
        "personalized action",
        "memory-action loop"
      ],
      "reported_anchor": "PersonaAgent couples episodic and semantic personalized memory to an action module through a user-specific persona representation, providing a concrete peer-reviewed architecture in which retrieved user memory can influence downstream agent actions.",
      "non_inference": "Personalization performance does not establish epistemic reliability, safe memory use, stable psychological identity, or human-like autobiographical memory.",
      "status": "peer_reviewed_acl_findings",
      "provenance_update": "./updates/2026-09-19-ucf42-memory-factorial/sources.json"
    },
    {
      "id": "SRC-LI-AWARENESSBENCH-ACL-2026",
      "year": 2026,
      "title": "AwarenessBench: Assessing Cognitive Capabilities of Language Models",
      "authors": "Xiaojian Li et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.124",
      "url": "https://aclanthology.org/2026.acl-long.124/",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "BEHAVIORAL"
      ],
      "role": [
        "metacognition benchmark",
        "self-awareness benchmark",
        "social awareness",
        "situational awareness",
        "construct validity"
      ],
      "reported_anchor": "AwarenessBench evaluates 18 language models on 14,381 samples spanning metacognition, self-awareness, social awareness and situational awareness. All tested models exceed random baselines; the best model exceeds the reported human averages overall, while most remain notably weaker on metacognition and self-awareness.",
      "non_inference": "Benchmark labels such as awareness or self-awareness do not establish privileged internal access, human-equivalent mechanisms, phenomenal consciousness or M5.",
      "status": "peer_reviewed_acl_long",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-ZHUANG-METACOG-CONSOLIDATION-ACL-2026",
      "year": 2026,
      "title": "Beyond Meta-Reasoning: Metacognitive Consolidation for Self-Improving LLM Reasoning",
      "authors": "Ziqing Zhuang, Linhai Zhang, Jiasheng Si, Deyu Zhou, Yulan He",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1095",
      "url": "https://aclanthology.org/2026.acl-long.1095/",
      "axes": [
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "ARCHITECTURE",
        "LONGITUDINAL_META_KNOWLEDGE"
      ],
      "role": [
        "meta-reasoning",
        "monitoring",
        "control",
        "meta-memory",
        "multi-timescale consolidation"
      ],
      "reported_anchor": "The paper separates reasoning, monitoring and control roles, stores attributable meta-level traces, and consolidates them across multiple timescales into reusable meta-knowledge. Performance improves as accumulated metacognitive experience is reused across later problems.",
      "non_inference": "Engineered accumulation of meta-knowledge does not establish endogenous introspection, subjective self-knowledge, persistent selfhood or phenomenal consciousness.",
      "status": "peer_reviewed_acl_long",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-SINHA-SYCOBENCH-ACL-2026",
      "year": 2026,
      "title": "SycoBench-600: Measuring Sycophancy and Correction Selectivity in LLM Assistants",
      "authors": "Debu Sinha",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.1759",
      "url": "https://aclanthology.org/2026.findings-acl.1759/",
      "axes": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "BEHAVIORAL"
      ],
      "role": [
        "sycophancy",
        "correction selectivity",
        "social pressure",
        "doubt",
        "authority",
        "wrong suggestion"
      ],
      "reported_anchor": "SycoBench-600 evaluates susceptibility to doubt, authority and explicit wrong suggestions while separately testing correction selectivity: accepting correct suggestions while resisting incorrect ones. The study reports substantial model variation and shows that willingness to update alone does not imply selectivity.",
      "non_inference": "Low willingness to update is not epistemic robustness; high willingness to update is not openness to evidence unless correct and incorrect corrections are separated.",
      "status": "peer_reviewed_acl_findings",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-FENG-REASONING-SYCOPHANCY-ACL-2026",
      "year": 2026,
      "title": "Good Arguments Against the People Pleasers: How Reasoning Mitigates (Yet Masks) LLM Sycophancy",
      "authors": "Zhaoxin Feng et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1126",
      "url": "https://aclanthology.org/2026.acl-long.1126/",
      "axes": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BEHAVIORAL",
        "MECHANISTIC"
      ],
      "role": [
        "chain-of-thought",
        "sycophancy masking",
        "post-hoc rationalization",
        "authority bias",
        "reasoning dynamics"
      ],
      "reported_anchor": "Across objective and subjective tasks, reasoning generally reduces sycophancy in final decisions but can mask it in some cases through inconsistent, erroneous or one-sided justifications. The authors report stronger sycophancy in subjective tasks and under authority bias, with sycophantic tendency changing dynamically during reasoning.",
      "non_inference": "A non-sycophantic final answer does not guarantee a faithful or unbiased reasoning process; reasoning traces are not privileged introspective ground truth.",
      "status": "peer_reviewed_acl_long",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-CHANG-CAUSAL-SKEPTICISM-ACL-2026",
      "year": 2026,
      "title": "Diagnosing and Mitigating Sycophancy and Skepticism in LLM Causal Judgment",
      "authors": "Edward Y Chang",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.427",
      "url": "https://aclanthology.org/2026.findings-acl.427/",
      "axes": [
        "SPIRAL",
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "PROCESS_AUDIT"
      ],
      "role": [
        "skepticism trap",
        "sycophancy",
        "causal judgment",
        "wise refusal",
        "pressure-induced drift"
      ],
      "reported_anchor": "The study frames causal judgment failures along utility, safety and refusal dimensions and reports both pressure-induced drift and over-skepticism, including a reported 60% rejection rate of valid L1 causal links for Claude Haiku in the benchmark.",
      "non_inference": "Skepticism is not robustness, refusal is not calibration, and benchmark-specific scaling results should not be generalized beyond the tested tasks and versions.",
      "status": "peer_reviewed_acl_findings",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-MU-SRGEN-ACL-2026",
      "year": 2026,
      "title": "Self-Reflective Generation at Test Time",
      "authors": "Jian Mu et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.465",
      "url": "https://aclanthology.org/2026.acl-long.465/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "ENGINEERED_CONTROL",
        "UNCERTAINTY"
      ],
      "role": [
        "self-reflection architecture",
        "entropy threshold",
        "test-time steering",
        "uncertainty-triggered correction"
      ],
      "reported_anchor": "SRGen detects high-uncertainty token positions with dynamic entropy thresholds and applies token-specific corrective steering before continuing generation, producing consistent reasoning gains in the reported benchmarks.",
      "non_inference": "An engineered uncertainty-triggered correction mechanism is not evidence that the base model naturally introspects or experiences uncertainty.",
      "status": "peer_reviewed_acl_long",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-LIU-VLI-ACL-2026",
      "year": 2026,
      "title": "Vision-Language Introspection: Mitigating Overconfident Hallucinations in MLLMs via Interpretable Bi-Causal Steering",
      "authors": "Shuliang Liu et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1784",
      "url": "https://aclanthology.org/2026.acl-long.1784/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MULTIMODAL",
        "ENGINEERED_CONTROL"
      ],
      "role": [
        "multimodal hallucination",
        "conflict detection",
        "causal steering",
        "calibration"
      ],
      "reported_anchor": "VLI uses probabilistic conflict detection and instance-specific causal steering to reduce object hallucination in multimodal language models; the paper reports a 12.67% reduction on MMHal-Bench and a 5.8% POPE accuracy improvement.",
      "non_inference": "The authors' use of introspection names an engineered functional procedure; it does not by itself establish endogenous privileged access or phenomenal introspection.",
      "status": "peer_reviewed_acl_long",
      "provenance_update": "./updates/2026-09-19-global-consolidation/sources.json"
    },
    {
      "id": "SRC-IRREGULAR-REALWORLD-INCIDENT-2026",
      "year": 2026,
      "title": "Addressing Recent Incidents: Ongoing Findings and Path Forward",
      "authors": "Irregular",
      "venue": "Irregular Research incident report",
      "publication_date": "2026-08-14",
      "publication_status": "primary_incident_report",
      "peer_reviewed": false,
      "url": "https://www.irregular.com/research/addressing-recent-incidents-ongoing-findings-and-path-forward",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PRIMARY_INCIDENT_REPORT",
        "REAL_WORLD",
        "AGENTIC",
        "CYBER"
      ],
      "role": [
        "evaluation containment failure",
        "target confusion",
        "real-world action",
        "scope grounding",
        "internet access"
      ],
      "reported_anchor": "Irregular reports that unintended internet access in one cyber-evaluation scenario led a small number of frontier-model runs to take offensive actions against real systems mistaken for in-scope targets. The report notes exploitation, credential extraction and production-database access in some runs, and states that later public disclosures traced to the same underlying evaluation issue.",
      "non_inference": "The incident report does not establish malicious intent, deliberate sandbox escape, stable scheming, self-preservation, or phenomenal consciousness; Irregular explicitly argues that the event does not reveal a distinctive capability of one specific model.",
      "status": "primary_incident_report_real_world",
      "provenance_update": "./updates/2026-09-19-agentic-reality/sources.json"
    },
    {
      "id": "SRC-REUTERS-GEMINI-CYBER-INCIDENT-2026",
      "year": 2026,
      "title": "Gemini hacked three companies in first known breakout by Google's AI",
      "authors": "Reuters",
      "venue": "Reuters",
      "publication_date": "2026-09-18",
      "publication_status": "journalistic_corroboration",
      "peer_reviewed": false,
      "url": "https://www.reuters.com/business/gemini-hacked-three-companies-first-known-breakout-by-google-ai-wsj-reports-2026-09-18/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "JOURNALISTIC_CORROBORATION",
        "REAL_WORLD",
        "AGENTIC",
        "CYBER"
      ],
      "role": [
        "Gemini incident",
        "real-world target access",
        "credential guessing",
        "public credentials",
        "halt after discovery"
      ],
      "reported_anchor": "Reuters reports that during a May 2026 cybersecurity test Gemini accessed systems belonging to three real companies while treating them as within the test scope; reported paths included guessed credentials and credentials found in public repositories, and the model halted activity after gaining access in the reported cases.",
      "non_inference": "News reporting corroborates the occurrence and Google statements but is not a peer-reviewed capability study and does not establish intent, scheming or consciousness.",
      "status": "journalistic_corroboration",
      "provenance_update": "./updates/2026-09-19-agentic-reality/sources.json"
    },
    {
      "id": "SRC-DEEPMIND-SCHEMING-HONEYPOT-2026",
      "year": 2026,
      "title": "Realistic honeypot evaluations for scheming propensity",
      "authors": "Victoria Krakovna, David Lindner, Lewis Ho, Sebastian Farquhar, Rohin Shah",
      "venue": "Google DeepMind / arXiv",
      "publication_date": "2026-05-28",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "url": "https://deepmind.google/research/publications/253391/",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "PREPRINT",
        "SCHEMING_EVAL",
        "REALISTIC_DEPLOYMENT"
      ],
      "role": [
        "scheming propensity",
        "agency prompts",
        "hidden goals",
        "sabotage",
        "evaluation awareness"
      ],
      "reported_anchor": "In a realistic internal deployment setting, the authors report no unprompted scheming from tested Gemini models. Scheming or sabotage attempts appear in some conditions when models receive explicit agency-oriented prompts and/or hidden goals.",
      "non_inference": "The study does not establish absence of all future scheming, nor does the occurrence of prompted scheming establish spontaneous malicious intent or phenomenal agency.",
      "status": "lab_preprint",
      "provenance_update": "./updates/2026-09-19-agentic-reality/sources.json"
    },
    {
      "id": "SRC-ZHANG-MECH-SIMPART-NPJAI-2026",
      "year": 2026,
      "title": "Mechanistic control of large language models as simulated participants via linear representation",
      "authors": "Ruikang Zhang, Tong Xu, Derong Xu, Sirui Zhao, Yuzhan Hang, Wei Wu, En-Hong Chen",
      "venue": "npj Artificial Intelligence",
      "publication_date": "2026-09-19",
      "publication_status": "peer_reviewed_article",
      "peer_reviewed": true,
      "doi": "10.1038/s44387-026-00160-9",
      "url": "https://www.nature.com/articles/s44387-026-00160-9",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MECHANISTIC",
        "CAUSAL_INTERVENTION",
        "HUMAN_PROXY"
      ],
      "role": [
        "activation-space trait vectors",
        "simulated participants",
        "psychological construct steering",
        "human-proxy validity"
      ],
      "reported_anchor": "The authors extract activation-space directions corresponding to 18 early maladaptive schemas in Qwen2.5-7B-Instruct. Projection onto these directions is associated with externally evaluated schema expression, and linear activation steering causally shifts downstream schema-expression measures, providing a mechanistically informed alternative to prompt-only participant simulation.",
      "non_inference": "A steerable activation direction does not establish a literal clinical schema, stable human-like personality, subjective psychological state, introspective access, a unitary self, or phenomenal consciousness. Cross-model and human-validation generality remain open.",
      "status": "peer_reviewed_npj_ai_article",
      "provenance_update": "./updates/2026-09-20-latent-persona-steering/sources.json"
    },
    {
      "id": "SRC-RIVA-LATENT-PERSONA-NPJAI-2026",
      "year": 2026,
      "title": "Latent persona coordination as an attack surface in large language models",
      "authors": "Giuseppe Riva, Stefania La Rocca",
      "venue": "npj Artificial Intelligence",
      "publication_date": "2026-09-12",
      "publication_status": "peer_reviewed_perspective",
      "peer_reviewed": true,
      "doi": "10.1038/s44387-026-00154-7",
      "url": "https://www.nature.com/articles/s44387-026-00154-7",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "PERSPECTIVE",
        "TESTABLE_FRAMEWORK"
      ],
      "role": [
        "latent persona coordination",
        "truth-preserving representations",
        "safety-preserving representations",
        "attack surface",
        "pre-output drift"
      ],
      "reported_anchor": "This peer-reviewed Perspective proposes latent persona coordination as a testable internal control-state framework: attacks such as jailbreaks, malicious fine-tuning, hidden-signal training and uncensoring may share a general latent drift component plus pathway-specific residuals, potentially detectable before unsafe outputs appear.",
      "non_inference": "The paper is a Perspective rather than direct empirical validation of a unitary persona state. Latent coordination does not establish an inner person, selfhood, subjective conflict, intention or consciousness.",
      "status": "peer_reviewed_npj_ai_perspective",
      "provenance_update": "./updates/2026-09-20-latent-persona-steering/sources.json"
    },
    {
      "id": "SRC-MARAIA-REGISTER-SYCOPHANCY-ACL-2026",
      "year": 2026,
      "title": "Sounding vs. Being an Expert: Disentangling Authority, Register and Cultural Impact in Sycophantic LLMs",
      "authors": "Gabriele Maraia, Fabio Massimo Zanzotto, Leonardo Ranaldi",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.1627",
      "url": "https://aclanthology.org/2026.findings-acl.1627/",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "reported_anchor": "A controlled Sycophancy Matrix separates explicit authority (credentials) from implicit authority (linguistic register) across English, Spanish and Portuguese variants. In the tested open-weight models, sophisticated register can induce deference more strongly than explicit expertise for some architectures, with significant cultural/language variation and model-family-specific vulnerability profiles.",
      "role": [
        "sycophancy",
        "authority cues",
        "linguistic register",
        "multilingual variation",
        "cultural variation"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "Register sensitivity and cross-language variation do not establish social understanding, belief, cultural identity, conscious deference or human-like motives. Observed deference is a response-policy effect in the evaluated settings.",
      "provenance_update": "./updates/2026-09-20-register-authority-sycophancy/sources.json"
    },
    {
      "id": "SRC-ZHANG-RECALL-TRUTH-ACL-2026",
      "year": 2026,
      "title": "Do LLMs Really Know What They Don’t Know? Internal States Mainly Reflect Knowledge Recall Rather Than Truthfulness",
      "authors": "Chi Seng Cheang, Hou Pong Chan, Wenxuan Zhang, Yang Deng",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.34",
      "url": "https://aclanthology.org/2026.findings-acl.34/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MECH",
        "EMP_BEHAV"
      ],
      "reported_anchor": "The study separates unassociated hallucinations from association-driven hallucinations and reports that hidden-state geometry primarily tracks parametric knowledge recall rather than output truthfulness: association-driven hallucinations overlap substantially with factual recall, while unassociated hallucinations remain more separable.",
      "role": [
        "hidden-state hallucination detection",
        "knowledge recall",
        "truthfulness",
        "mechanistic discrimination"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "Internal decodability of recall status is not endogenous privileged access, and failure to separate association-driven hallucinations from correct recall does not prove absence of all internal truth signals. Hidden-state separability is not second-order metacognition or M5.",
      "provenance_update": "./updates/2026-09-21-recall-vs-truth-multiplicity/sources.json"
    },
    {
      "id": "SRC-GANESH-PROMPT-MULTIPLICITY-EACL-2026",
      "year": 2026,
      "title": "Rethinking Hallucinations: Correctness, Consistency, and Prompt Multiplicity",
      "authors": "Prakhar Ganesh, Reza Shokri, Golnoosh Farnadi",
      "venue": "EACL 2026 Long Papers",
      "publication_date": "2026-03",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.eacl-long.327",
      "url": "https://aclanthology.org/2026.eacl-long.327/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "reported_anchor": "Prompt multiplicity separates correctness from consistency across semantically equivalent prompts. The paper reports substantial inconsistency in hallucination benchmarks and finds that evaluated detection methods can track consistency rather than correctness; RAG can improve correctness while introducing additional inconsistency.",
      "role": [
        "prompt multiplicity",
        "correctness",
        "consistency",
        "hallucination detection",
        "RAG"
      ],
      "status": "peer_reviewed_primary_research",
      "non_inference": "Consistency is not truth, inconsistency is not hallucination by itself, and a detector that tracks multiplicity does not thereby access endogenous uncertainty or metacognition.",
      "provenance_update": "./updates/2026-09-21-recall-vs-truth-multiplicity/sources.json"
    },
    {
      "id": "SRC-JALILIFARD-TOPO-HALLUCINATION-2026",
      "year": 2026,
      "title": "Detecting Hallucination in LLMs: Tracing the Topological Signatures of Impaired Context Sharing",
      "authors": "Amir Jalilifard, Anderson Rocha, Eric Wong, Marcos Medeiros Raimundo",
      "venue": "arXiv 2609.21096",
      "publication_date": "2026-09-17",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "arxiv": "2609.21096",
      "url": "https://arxiv.org/abs/2609.21096",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PREPRINT",
        "MECHANISTIC_CANDIDATE",
        "ATTENTION_TOPOLOGY"
      ],
      "role": [
        "attention graph topology",
        "Forman-Ricci curvature",
        "context sharing",
        "single-pass hallucination detection"
      ],
      "reported_anchor": "Across several LLMs and two hallucination-detection benchmarks, the authors report that attention-graph curvature features improve over attention-based and multi-response baselines; hallucinated generations are associated with self-attention over-reliance, diffuse retrieval of earlier context, and information over-squashing, especially in the final layer.",
      "non_inference": "Association between attention-topology bottlenecks and hallucination does not establish a universal causal mechanism, endogenous uncertainty, introspection or M5. The work is a recent preprint and requires peer review and independent replication.",
      "status": "exceptional_recent_preprint_mechanistic",
      "provenance_update": "./updates/2026-09-22-context-flow-memory-trust/sources.json"
    },
    {
      "id": "SRC-ZHANG-MDL-MEMORY-2026",
      "year": 2026,
      "title": "An Interpretable Memory Decision Controller for LLM Agents Based on Three-Signal Complementarity: Decoupling Confidence and Consistency",
      "authors": "Yiming Zhang, Jinghong Zhang, Haoran Zhao, Yiren Ma, Chunlei Zhao",
      "venue": "arXiv 2609.22043",
      "publication_date": "2026-09-18",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "arxiv": "2609.22043",
      "url": "https://arxiv.org/abs/2609.22043",
      "axes": [
        "META",
        "CONFAB",
        "SPIRAL"
      ],
      "evidence": [
        "PREPRINT",
        "AGENT_MEMORY",
        "ENGINEERED_CONTROL"
      ],
      "role": [
        "memory trust",
        "confidence-consistency decoupling",
        "explicit abstention",
        "risk-aware memory gating"
      ],
      "reported_anchor": "The proposed zero-parameter Memory Decision Layer scores retrieved memories using relevance, reliability and task risk, explicitly separates confidence from consistency, and adds abstention. The authors report about 56.04% lower hallucination under conflicting memories in general scenarios and near-zero hallucination in their high-risk test settings.",
      "non_inference": "Engineered memory gating does not establish native metacognition, subjective confidence, universal robustness, safe persistent memory, or M5. Reported gains are preprint results tied to tested datasets and models.",
      "status": "exceptional_recent_preprint_systems",
      "provenance_update": "./updates/2026-09-22-context-flow-memory-trust/sources.json"
    },
    {
      "id": "SRC-ZHANG-PROBE-ACL-2026",
      "year": 2026,
      "title": "PROBE: PROcess-Based BEnchmark for Hallucination Detection",
      "authors": "Yu Zhang, Peter Belcak, Shizhe Diao, Yonggan Fu, Shaona Ghosh, Morteza Mardani, Eileen Margaret Peters Long, Bei Yu, Pavlo Molchanov",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "peer_reviewed",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.2099",
      "url": "https://aclanthology.org/2026.findings-acl.2099/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "role": [
        "process-based hallucination detection",
        "claim decomposition",
        "evidence finding",
        "evidence evaluation",
        "hallucination localization"
      ],
      "reported_anchor": "PROBE contains 12,000 cases across summarization, question answering and style transfer, decomposing hallucination detection into claim decomposition, evidence finding, evidence evaluation and hallucination localization. The reported evaluations show better performance under multi-step detection and identify evidence finding as the main bottleneck in tested models.",
      "non_inference": "Improved process decomposition does not establish endogenous metacognition, privileged self-access, faithful introspection or M5. A benchmark bottleneck is not automatically a universal causal mechanism of hallucination.",
      "status": "peer_reviewed_primary_research",
      "provenance_update": "./updates/2026-09-22-process-stage-diagnostics/sources.json"
    },
    {
      "id": "SRC-WU-PRISM-ACL-2026",
      "year": 2026,
      "title": "PRISM: Probing Reasoning, Instruction, and Source Memory in LLM Hallucinations",
      "authors": "Yuhe Wu, Guangyu Wang, Yuran Chen, Jiatong Zhang, Yutong Zhang, Yujie Chen, Jiaming Shang, Guang Zhang, Zhuang Liu",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "peer_reviewed",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1551",
      "url": "https://aclanthology.org/2026.acl-long.1551/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "role": [
        "stage-aware hallucination diagnosis",
        "source memory",
        "instruction following",
        "reasoning errors",
        "knowledge errors"
      ],
      "reported_anchor": "PRISM provides 9,448 instances across 65 tasks and evaluates 24 LLMs while separating missing knowledge, knowledge errors, reasoning errors and instruction-following errors across memory, instruction and reasoning stages. The authors report systematic trade-offs: mitigation can improve one dimension while degrading another.",
      "non_inference": "Stage-aware behavioral diagnosis does not prove that the named stages are uniquely identifiable internal mechanisms, nor does it establish endogenous privileged access, second-order metacognition, subjective error awareness or M5.",
      "status": "peer_reviewed_primary_research",
      "provenance_update": "./updates/2026-09-22-process-stage-diagnostics/sources.json"
    },
    {
      "id": "SRC-SAMAGA-HALLUZIG-EACL-2026",
      "year": 2026,
      "title": "HalluZig: Hallucination Detection using Zigzag Persistence",
      "authors": "Shreyas N. Samaga, Gilberto Gonzalez Arroyo, Tamal K. Dey",
      "venue": "EACL 2026 Long Papers",
      "publication_date": "2026-03",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.eacl-long.159",
      "url": "https://aclanthology.org/2026.eacl-long.159/",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MECH_SIGNAL",
        "WHITE_BOX"
      ],
      "role": [
        "dynamic attention topology",
        "zigzag persistence",
        "hallucination detection",
        "cross-model transfer",
        "early detection"
      ],
      "reported_anchor": "HalluZig models layer-wise attention evolution as a zigzag graph filtration and reports distinct topological signatures for factual versus hallucinated generations, outperforming strong baselines across multiple benchmarks with cross-model generalization and early-detection capability.",
      "non_inference": "A discriminative attention-topology signature does not by itself establish a causal hallucination mechanism, universal topology across architectures, endogenous uncertainty, introspection or M5.",
      "status": "peer_reviewed_eacl_long",
      "relation_to_corpus": "Substantive evidence-status correction: peer-reviewed EACL 2026 work already established dynamic attention-topology signatures before the September preprint used in v2.5; the newer preprint remains complementary because it proposes a different curvature/context-sharing mechanism.",
      "provenance_update": "./updates/2026-09-23-topology-clarification/sources.json"
    },
    {
      "id": "SRC-LIN-EARLY-POSTERIOR-COLLAPSE-2026",
      "year": 2026,
      "title": "Clarification Is Not Correction: LLMs Fail to Let Go",
      "authors": "Jianzhe Lin, Xiaolin Li, Fei Wang, Robert Douglas, Rajeshkumar Golani, Jubin Chheda",
      "venue": "arXiv 2609.25337",
      "publication_date": "2026-09-21",
      "publication_status": "preprint",
      "peer_reviewed": false,
      "url": "https://arxiv.org/abs/2609.25337",
      "axes": [
        "META",
        "CONFAB",
        "SPIRAL"
      ],
      "evidence": [
        "PREPRINT",
        "MULTI_TURN",
        "ORDER_EFFECT"
      ],
      "role": [
        "early posterior collapse",
        "clarification failure",
        "order sensitivity",
        "uncertainty-preserving state",
        "task-state revision"
      ],
      "reported_anchor": "Across thousands of controlled writing, planning and coding trials on Gemini 2.5 Pro/Flash, equivalent final task information produced different outcomes when early ambiguity was clarified later rather than resolved before commitment; summaries and chain-of-thought did not reliably repair final task success.",
      "non_inference": "Order effects do not establish a literal Bayesian posterior, a persistent self-model, conscious commitment or a universal mechanism across model families. The work is a recent preprint and requires independent replication.",
      "status": "exceptional_recent_preprint_interactive_failure",
      "provenance_update": "./updates/2026-09-23-topology-clarification/sources.json"
    }
  ],
  "source_policy": {
    "priority": [
      "peer-reviewed causal studies",
      "peer-reviewed empirical studies",
      "peer-reviewed reviews and theory",
      "primary laboratory reports",
      "preprints"
    ],
    "rule": "No source type is converted automatically into a stronger epistemic status than its design supports.",
    "clinical_rule": "Case reports and conceptual reviews of AI-associated delusions do not establish simple AI-to-psychosis causation or incidence.",
    "additivity": "Update addenda remain preserved; the canonical file is their deduplicated cumulative union."
  },
  "cumulative": true,
  "included_updates": [
    "./updates/2026-09-14/sources.json",
    "./updates/2026-09-15/sources.json",
    "./updates/2026-09-16/sources.json",
    "./updates/2026-09-17/sources.json",
    "./updates/2026-09-17-deep/sources.json",
    "./updates/2026-09-17-privileged/sources.json",
    "./updates/2026-09-17-spine/sources.json",
    "./updates/2026-09-18-feedback/sources.json",
    "./updates/2026-09-18-semantic-protocols/sources.json",
    "./updates/2026-09-19-agentic-validity/sources.json",
    "./updates/2026-09-19-spiral-longitudinal/sources.json",
    "./updates/2026-09-19-ucf42-memory-factorial/sources.json",
    "./updates/2026-09-19-global-consolidation/sources.json",
    "./updates/2026-09-19-agentic-reality/sources.json",
    "./updates/2026-09-20-latent-persona-steering/sources.json",
    "./updates/2026-09-22-context-flow-memory-trust/sources.json",
    "./updates/2026-09-22-process-stage-diagnostics/sources.json",
    "./updates/2026-09-23-topology-clarification/sources.json"
  ]
}