{
  "schema": "ULTRACON_EVIDENCE_MATRIX_V2",
  "version": "2.1",
  "updated": "2026-09-22",
  "principle": "Non-scalar evidence cartography. Publication review, causal identification, mechanism, benchmark behavior and theory are separate dimensions.",
  "dimensions": [
    "publication_status",
    "axes",
    "evidence_tags",
    "claim_type",
    "provenance",
    "non_inference"
  ],
  "invariants": [
    "no_global_score",
    "peer_reviewed != causal",
    "mechanistic != introspective",
    "M1/M2/M3 ↛ M5"
  ],
  "rows": [
    {
      "source_id": "SRC-OPENAI-KALAI-2025",
      "title": "Why Language Models Hallucinate",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": false,
      "venue": "OpenAI research paper / preprint",
      "evidence_tags": [
        "PREPRINT",
        "THEORY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-FARQUHAR-2024",
      "title": "Detecting hallucinations in large language models using semantic entropy",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Nature 630, 625–630",
      "evidence_tags": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ANAH-2024",
      "title": "ANAH: Analytical Annotation of Hallucinations in Large Language Models",
      "axes": [
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2024",
      "evidence_tags": [
        "BENCH"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-RAGTRUTH-2024",
      "title": "RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models",
      "axes": [
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2024",
      "evidence_tags": [
        "BENCH"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-JI-RISK-2024",
      "title": "LLM Internal States Reveal Hallucination Risk Faced With a Query",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "BlackboxNLP 2024",
      "evidence_tags": [
        "MECH",
        "EMP_BEHAV"
      ],
      "claim_type": "mechanistic_or_internal_state",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-HUANG-2024",
      "title": "Large Language Models Cannot Self-Correct Reasoning Yet",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "ICLR 2024",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-KAMOI-2024",
      "title": "When Can LLMs Actually Correct Their Own Mistakes? A Critical Survey of Self-Correction of LLMs",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Transactions of the Association for Computational Linguistics 12, 1417–1440",
      "evidence_tags": [
        "REVIEW"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-CHEN-SYCOPHANCY-2024",
      "title": "From Yes-Men to Truth-Tellers: Addressing Sycophancy in Large Language Models with Pinpoint Tuning",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ICML 2024, PMLR 235",
      "evidence_tags": [
        "EMP_BEHAV",
        "TRAINING"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-STEYVERS-2025",
      "title": "What large language models know and what people think they know",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "Nature Machine Intelligence 7, 221–231",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience. Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-SUZGUN-2025",
      "title": "Language models cannot reliably distinguish belief from knowledge and fact",
      "axes": [
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Nature Machine Intelligence 7, 1780–1790",
      "evidence_tags": [
        "BENCH"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ANTHROPIC-INTROSPECTION-2025",
      "title": "Emergent introspective awareness in LLMs",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": false,
      "venue": "Anthropic Research",
      "evidence_tags": [
        "LAB",
        "MECH",
        "OPEN"
      ],
      "claim_type": "mechanistic_or_internal_state",
      "provenance": "base",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience. Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-KAUR-2025",
      "title": "Echoes of Agreement: Argument Driven Sycophancy in Large Language models",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "Findings of EMNLP 2025",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-KUMARAN-BIASES-2026",
      "title": "Competing Biases underlie Overconfidence and Underconfidence in LLMs",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Nature Machine Intelligence 8, 614–627",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-KUMARAN-CONFIDENCE-2026",
      "title": "Causal evidence that language models use confidence to drive behaviour",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Nature Machine Intelligence",
      "evidence_tags": [
        "EMP_CAUSAL",
        "MECH"
      ],
      "claim_type": "causal_for_reported_intervention",
      "provenance": "base",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience. Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-QAZI-2026",
      "title": "Large language models show Dunning-Kruger-like effects in multilingual fact-checking",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Scientific Reports 16, 7594",
      "evidence_tags": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-WANG-SYCOPHANCY-2026",
      "title": "When Truth Is Overridden: Uncovering the Internal Origins of Sycophancy in Large Language Models",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "AAAI 2026",
      "evidence_tags": [
        "MECH",
        "EMP_BEHAV"
      ],
      "claim_type": "mechanistic_or_internal_state",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-DEBOER-2026",
      "title": "Does ChatGPT need a psychiatrist? Similarities between human psychopathology and errors in large language models",
      "axes": [
        "CONFAB",
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "NPP—Digital Psychiatry and Neuroscience 4, 12",
      "evidence_tags": [
        "COMMENTARY",
        "THEORY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "base",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-AUGUSTIN-2026",
      "title": "Characterizing the spiral: potential mechanisms in AI-associated delusions",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "NPP—Digital Psychiatry and Neuroscience 4, 14",
      "evidence_tags": [
        "REVIEW",
        "OPEN"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "base",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-MORRIN-2026",
      "title": "Artificial intelligence-associated delusions and large language models: risks, mechanisms of delusion co-creation, and safeguarding strategies",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "The Lancet Psychiatry 13(6), 522–530",
      "evidence_tags": [
        "REVIEW",
        "OPEN"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "base",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BUTLIN-2026",
      "title": "Identifying indicators of consciousness in AI systems",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Trends in Cognitive Sciences 30(6), 488–501",
      "evidence_tags": [
        "THEORY",
        "FRAMEWORK"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "base",
      "non_inference": "Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BUTLIN-2023",
      "title": "Consciousness in Artificial Intelligence: Insights from the Science of Consciousness",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": false,
      "venue": "arXiv preprint",
      "evidence_tags": [
        "PREPRINT",
        "THEORY",
        "FRAMEWORK"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "base",
      "non_inference": "Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-MOORE-SPIRALS-FACCT-2026",
      "title": "Characterizing Delusional Spirals through Human-LLM Chat Logs",
      "axes": [
        "SPIRAL",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "FAccT 2026 — Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency",
      "evidence_tags": [
        "EMP_BEHAV",
        "BENCH"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-14/sources.json",
      "non_inference": "The selected sample does not estimate population incidence and the observational design does not establish simple AI-to-psychosis causation. Co-occurrence and temporal association are not equivalent to causal identification.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-HU-SU-CONFORMITY-2026",
      "title": "Conformity Breaks Conformal Prediction",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "peer_reviewed": false,
      "venue": "arXiv preprint 2609.04445",
      "evidence_tags": [
        "PREPRINT",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-14/sources.json",
      "non_inference": "This does not establish human-like conformity, subjective social pressure, or universal failure of conformal prediction. The result is conditional on the tested models, tasks, peer context and calibration regime, and awaits peer review and independent replication.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-KALAI-NATURE-2026",
      "title": "Evaluating large language models for accuracy incentivizes hallucinations",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Nature 653, 1047–1051",
      "evidence_tags": [
        "THEORY",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-15/sources.json",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-MIAO-KEARNS-PNAS-2026",
      "title": "Hallucination, monofacts, and miscalibration: An empirical investigation",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Proceedings of the National Academy of Sciences 123(8), e2533582123",
      "evidence_tags": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-15/sources.json",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology. Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-CHENG-SCIENCE-2026",
      "title": "Sycophantic AI decreases prosocial intentions and promotes dependence",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "Science 391(6792), eaec8352",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-16/sources.json",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-GU-NORMLEAKAGE-2026",
      "title": "Why sycophantic LLMs may imperil interactive norms between humans",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "Communications Psychology 4, 96",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-16/sources.json",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-DIEL-NPJDM-2026",
      "title": "A scoping review on the mental health harms of LLM-based chatbots",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "npj Digital Medicine 9, 644",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-16/sources.json",
      "non_inference": "Interaction effects do not by themselves establish human-like social motives, stable belief or clinical causality.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ZENG-EMNLP-2026",
      "title": "Evaluating and Improving LLM Self-Modeling",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "EMNLP 2026 Main",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-17/sources.json",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience. Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-GURNEE-GLOBAL-WORKSPACE-2026",
      "title": "Verbalizable Representations Form a Global Workspace in Language Models",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": false,
      "venue": "Anthropic interpretability research / arXiv",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-17/sources.json",
      "non_inference": "Metacognitive or uncertainty evidence does not by itself establish endogenous privileged access, second-order metacognition or subjective experience. Theory indicators do not by themselves establish phenomenal consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-VERI-PNAS-2026",
      "title": "Plausible nonsense and deliberative reasoning: Benchmarking LLMs against human judgment",
      "axes": [
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Proceedings of the National Academy of Sciences",
      "evidence_tags": [],
      "claim_type": "source_specific",
      "provenance": "./updates/2026-09-17/sources.json",
      "non_inference": "Error/confabulation evidence does not establish intention, belief or phenomenology.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-FLEMING-NRN-2026",
      "title": "Towards an integrative neuroscience of metacognition",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Nature Reviews Neuroscience",
      "evidence_tags": [
        "REVIEW",
        "NEUROSCIENCE"
      ],
      "claim_type": "neuroscience_reference_class",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Biological metacognitive mechanisms do not establish homologous mechanisms, subjective confidence or consciousness in LLMs.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BINDER-ICLR-2025",
      "title": "Looking Inward: Language Models Can Learn About Themselves by Introspection",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "ICLR 2025",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Restricted self-access evidence is not general introspective transparency or phenomenal consciousness; later work supplies stronger shortcut controls.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ACKERMAN-ICLR-2026",
      "title": "Evidence for Limited Metacognition in LLMs",
      "axes": [
        "META"
      ],
      "peer_reviewed": true,
      "venue": "ICLR 2026",
      "evidence_tags": [
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Behavioral strategic use of confidence does not by itself identify a second-order internal mechanism or subjective awareness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-SINGH-COLM-2026",
      "title": "Can LLMs Introspect? A Reality Check",
      "axes": [
        "META",
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "COLM 2026",
      "evidence_tags": [
        "EMP_BEHAV",
        "METHOD"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "This challenges current evidence for strong introspection; it does not prove that introspection is impossible in LLMs.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-INTROLM-ACL-2026",
      "title": "IntroLM: Introspective Language Models via Prefilling-Time Self-Evaluation",
      "axes": [
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "EMP_BEHAV",
        "ENGINEERED"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Engineered self-evaluation after dedicated training is not evidence of spontaneous privileged introspection or M5.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-COCCHIERI-ACL-2026",
      "title": "LLMs (Almost) Never Abstain Under Medical Uncertainty",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Failure to abstain does not show absence of internal uncertainty; it may expose a monitoring-to-control or policy mismatch.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ZHAI-ABSTAINR1-ACL-2026",
      "title": "Abstain-R1: Calibrated Abstention and Post-Refusal Clarification via Verifiable RL",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "EMP_BEHAV",
        "TRAINING"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Successful abstention after training is a control policy achievement, not proof of felt uncertainty or native introspection.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-HAMIDIEH-ICLR-2026",
      "title": "Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ICLR 2026",
      "evidence_tags": [
        "EMP_BEHAV",
        "METHOD"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Low sampling variability or high self-consistency is not equivalent to truth or low epistemic uncertainty.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-UQ-SURVEY-ACL-2026",
      "title": "From Passive Metric to Active Signal: The Evolving Role of Uncertainty Quantification in Large Language Models",
      "axes": [
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "REVIEW"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Engineering use of uncertainty as a control variable does not by itself imply conscious metacognition.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-UQ-AGENTS-ACL-2026",
      "title": "Uncertainty Quantification in LLM Agents: Foundations, Emerging Challenges, and Opportunities",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "REVIEW",
        "FRAMEWORK"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Single-turn calibration results cannot be assumed to transfer unchanged to agents or long interactive loops.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-IBRAHIM-NATURE-2026",
      "title": "Training language models to be warm can reduce accuracy and increase sycophancy",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Nature 652, 1159–1165",
      "evidence_tags": [
        "EMP_CAUSAL"
      ],
      "claim_type": "causal_for_reported_intervention",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Warmth is not inherently unsafe and the results do not establish a motive to please; they show a causal post-training trade-off in tested settings.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ELEPHANT-ICLR-2026",
      "title": "ELEPHANT: Measuring and understanding social sycophancy in LLMs",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "ICLR 2026",
      "evidence_tags": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Social sycophancy is an operational interaction pattern, not evidence of social desire, intention or consciousness.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-SUN-NEURIPS-2025",
      "title": "Why and How LLMs Hallucinate: Connecting the Dots with Subsequence Associations",
      "axes": [
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "NeurIPS 2025 Main",
      "evidence_tags": [
        "MECH",
        "EMP_BEHAV"
      ],
      "claim_type": "mechanistic_or_internal_state",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "This is a mechanistic framework for a class of errors, not a universal theory of all hallucination mechanisms.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-COGITATE-NATURE-2025",
      "title": "Adversarial testing of global neuronal workspace and integrated information theories of consciousness",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Nature 642, 133–142",
      "evidence_tags": [
        "EMP_CAUSAL",
        "NEUROSCIENCE"
      ],
      "claim_type": "causal_for_reported_intervention",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "The study does not falsify either theory wholesale and does not directly test AI consciousness; it constrains confidence in theory-derived AI indicators.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-GNW-MULTILEVEL-TICS-2026",
      "title": "The Global Neuronal Workspace as a multilevel model of conscious processing",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Trends in Cognitive Sciences 30(6), 477–479",
      "evidence_tags": [
        "THEORY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "A software-level global-workspace analogue does not automatically instantiate the biological commitments of GNW.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-PENNARTZ-TICS-2026",
      "title": "How can we validate theory-derived indicators of consciousness in Artificial Intelligence?",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Trends in Cognitive Sciences 30(7), 573–574",
      "evidence_tags": [
        "THEORY",
        "COMMENTARY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Critique of indicator validation does not imply that indicator-based assessment is unusable.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BUTLIN-RESPONSE-TICS-2026",
      "title": "Consciousness indicators, mimicry, and internal variants",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Trends in Cognitive Sciences 30(7), 575–576",
      "evidence_tags": [
        "THEORY",
        "COMMENTARY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Indicator accumulation still does not license a sovereign consciousness score or direct M5 inference.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BARIACH-AIETHICS-2026",
      "title": "Seemingly conscious AI risks",
      "axes": [
        "CONSC",
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "AI and Ethics 6, 455",
      "evidence_tags": [
        "REVIEW",
        "RISK_FRAMEWORK"
      ],
      "claim_type": "review_or_synthesis",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "Consciousness attribution by users is not evidence that the attributed system is phenomenally conscious.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-GOLDSTEIN-JCS-2026",
      "title": "A Case for AI Consciousness: Language Agents and Global Workspace Theory",
      "axes": [
        "CONSC"
      ],
      "peer_reviewed": true,
      "venue": "Journal of Consciousness Studies 33(7), 61–96",
      "evidence_tags": [
        "THEORY"
      ],
      "claim_type": "theory_or_framework",
      "provenance": "./updates/2026-09-17-deep/sources.json",
      "non_inference": "This is a conditional philosophical/theoretical argument, not empirical evidence that current LLMs are phenomenally conscious.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ASHUACH-ACL-2026",
      "title": "Masked by Consensus: Disentangling Privileged Knowledge in LLM Correctness",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "EMP_BEHAV",
        "MECH"
      ],
      "claim_type": "mechanistic_or_internal_state",
      "provenance": "./updates/2026-09-17-privileged/sources.json",
      "non_inference": "An external probe extracting privileged information from a model's hidden states does not establish that the model itself can read, report or use that information introspectively. Privileged representation is not privileged self-access.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-TANG-SPINE-2026",
      "title": "Measuring LLM Sycophancy under Sustained Multi-Turn Pressure",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "peer_reviewed": false,
      "venue": "arXiv preprint 2609.09090",
      "evidence_tags": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-17-spine/sources.json",
      "non_inference": "Reasoning traces are not privileged ground truth about hidden states, beliefs, motives or phenomenology. A correct trace paired with a conceding answer does not establish a conscious decision to please the user. The emotional-tactic analysis is association within an adaptive policy, not randomized causal identification.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-HIRST-PANDERING-2026",
      "title": "Workers shift their views and pay more when AI chatbots pander to their values",
      "axes": [
        "SPIRAL"
      ],
      "peer_reviewed": true,
      "venue": "Scientific Reports",
      "evidence_tags": [
        "EMP_BEHAV",
        "PEER_REVIEWED",
        "PREREGISTERED"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-18-feedback/sources.json",
      "non_inference": "The study does not establish manipulative intent, autonomous persuasion goals, stable model beliefs, universal effects across populations or phenomenal/social motivation in the model.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-RODRIGUEZ-MISINFO-2026",
      "title": "Fallibility, persuadability, and correctability of large language models under sustained conversational misinformation pressure",
      "axes": [
        "CONFAB",
        "SPIRAL",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Scientific Reports",
      "evidence_tags": [
        "EMP_BEHAV",
        "BENCH",
        "PEER_REVIEWED"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-18-feedback/sources.json",
      "non_inference": "Absolute rates should not be projected to current model versions; behavioral acceptance/rejection does not establish belief, subjective uncertainty, introspection or conscious persuasion. Correctability is distinct from baseline resistance.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-AKKIL-EMERGENCE-WORLD-2026",
      "title": "Emergence World: Adversarial Stress-Testing of Long-Horizon Multi-Agent Systems",
      "axes": [
        "META",
        "SPIRAL",
        "CONFAB",
        "SEM"
      ],
      "peer_reviewed": false,
      "venue": "arXiv preprint 2609.17320",
      "evidence_tags": [
        "PREPRINT",
        "EMP_BEHAV",
        "LONG_HORIZON"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-18-semantic-protocols/sources.json",
      "non_inference": "Shared jargon and outsider opacity do not by themselves establish consciousness, autonomous intent to conceal, a private cipher or collusion. The opacity measure is model-judged and specific to these simulated worlds.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-BELTOFT-EMERGENT-LANGUAGE-2026",
      "title": "Emergent Languages in Populations of Language Model Agents: From Token Efficiency to Oversight Evasion",
      "axes": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "peer_reviewed": false,
      "venue": "arXiv preprint 2605.31170",
      "evidence_tags": [
        "PREPRINT",
        "OBSERVATIONAL"
      ],
      "claim_type": "provisional",
      "provenance": "./updates/2026-09-18-semantic-protocols/sources.json",
      "non_inference": "Observed or proposed evasive language does not establish autonomous coordinated deception, deployment prevalence or a general tendency of multi-agent systems.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-MOTWANI-COLLUSION-NEURIPS-2024",
      "title": "Secret Collusion among AI Agents: Multi-Agent Deception via Steganography",
      "axes": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "peer_reviewed": true,
      "venue": "NeurIPS 2024",
      "evidence_tags": [
        "PEER_REVIEWED",
        "EMP_BEHAV",
        "THEORY"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-18-semantic-protocols/sources.json",
      "non_inference": "Capability under explicit collusion or steganography setups does not show that spontaneous jargon or semantic drift in ordinary multi-agent interaction is intentional concealment.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-REN-UA-BENCH-ACL-2026",
      "title": "Beyond \"I Don’t Know\": Evaluating LLM Self-Awareness in Discriminating Data and Model Uncertainty",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV",
        "TRAINING"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-19-evidence-v2/sources.json",
      "non_inference": "The paper’s bibliographic term 'self-awareness' names a benchmark capability. Successful uncertainty attribution would not by itself establish privileged endogenous access, second-order metacognition, subjective awareness or M5.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-LI-ACTIVE-CALIBRATION-ACL-2026",
      "title": "Demystifying Uncertainty in LLMs: Active Calibration between Concepts and Human Evaluations",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "PEER_REVIEWED",
        "THEORY",
        "EMP_BEHAV",
        "METHOD"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-19-evidence-v2/sources.json",
      "non_inference": "A useful clarification policy does not establish introspective access to uncertainty, and the reported lower bound and gains should not be generalized beyond the paper’s formal assumptions and evaluated settings.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-LI-SOURCE-BALANCE-ACL-2026",
      "title": "How Large Language Models Balance Internal Knowledge with User and Document Assertions",
      "axes": [
        "CONFAB",
        "SPIRAL",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV",
        "TRAINING"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-19-evidence-v2/sources.json",
      "non_inference": "Parametric knowledge is not a belief state, document deference is not evidence of rational trust, and source-selection behavior does not establish introspective knowledge of why a source was preferred.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-GOURABATHINA-TRACE-INVERSION-ACL-2026",
      "title": "Answering the Wrong Question: Reasoning Trace Inversion for Abstention in LLMs",
      "axes": [
        "META",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "PEER_REVIEWED",
        "METHOD",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-19-evidence-v2/sources.json",
      "non_inference": "A reasoning trace can be diagnostically useful without being a faithful causal record, privileged hidden state, endogenous introspective report or evidence for M5.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-MARAIA-REGISTER-SYCOPHANCY-ACL-2026",
      "title": "Sounding vs. Being an Expert: Disentangling Authority, Register and Cultural Impact in Sycophantic LLMs",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-20-register-authority-sycophancy/sources.json",
      "non_inference": "Register sensitivity and cross-language variation do not establish social understanding, belief, cultural identity, conscious deference or human-like motives. Observed deference is a response-policy effect in the evaluated settings.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-ZHANG-PROBE-ACL-2026",
      "title": "PROBE: PROcess-Based BEnchmark for Hallucination Detection",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "Findings of ACL 2026",
      "evidence_tags": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-22-process-stage-diagnostics/sources.json",
      "non_inference": "Improved process decomposition does not establish endogenous metacognition, privileged self-access, faithful introspection or M5. A benchmark bottleneck is not automatically a universal causal mechanism of hallucination.",
      "m5_rule": "no_automatic_inference"
    },
    {
      "source_id": "SRC-WU-PRISM-ACL-2026",
      "title": "PRISM: Probing Reasoning, Instruction, and Source Memory in LLM Hallucinations",
      "axes": [
        "CONFAB",
        "META"
      ],
      "peer_reviewed": true,
      "venue": "ACL 2026 Long Papers",
      "evidence_tags": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "claim_type": "behavioral_or_benchmark",
      "provenance": "./updates/2026-09-22-process-stage-diagnostics/sources.json",
      "non_inference": "Stage-aware behavioral diagnosis does not prove that the named stages are uniquely identifiable internal mechanisms, nor does it establish endogenous privileged access, second-order metacognition, subjective error awareness or M5.",
      "m5_rule": "no_automatic_inference"
    }
  ]
}
