{
  "schema": "ULTRACON_AI_EPISTEMIC_UPDATE_SOURCES_V1",
  "date": "2026-09-19",
  "sources": [
    {
      "id": "SRC-LI-AWARENESSBENCH-ACL-2026",
      "year": 2026,
      "title": "AwarenessBench: Assessing Cognitive Capabilities of Language Models",
      "authors": "Xiaojian Li et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.124",
      "url": "https://aclanthology.org/2026.acl-long.124/",
      "axes": [
        "META",
        "CONSC"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "BEHAVIORAL"
      ],
      "role": [
        "metacognition benchmark",
        "self-awareness benchmark",
        "social awareness",
        "situational awareness",
        "construct validity"
      ],
      "reported_anchor": "AwarenessBench evaluates 18 language models on 14,381 samples spanning metacognition, self-awareness, social awareness and situational awareness. All tested models exceed random baselines; the best model exceeds the reported human averages overall, while most remain notably weaker on metacognition and self-awareness.",
      "non_inference": "Benchmark labels such as awareness or self-awareness do not establish privileged internal access, human-equivalent mechanisms, phenomenal consciousness or M5.",
      "status": "peer_reviewed_acl_long"
    },
    {
      "id": "SRC-ZHUANG-METACOG-CONSOLIDATION-ACL-2026",
      "year": 2026,
      "title": "Beyond Meta-Reasoning: Metacognitive Consolidation for Self-Improving LLM Reasoning",
      "authors": "Ziqing Zhuang, Linhai Zhang, Jiasheng Si, Deyu Zhou, Yulan He",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1095",
      "url": "https://aclanthology.org/2026.acl-long.1095/",
      "axes": [
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "ARCHITECTURE",
        "LONGITUDINAL_META_KNOWLEDGE"
      ],
      "role": [
        "meta-reasoning",
        "monitoring",
        "control",
        "meta-memory",
        "multi-timescale consolidation"
      ],
      "reported_anchor": "The paper separates reasoning, monitoring and control roles, stores attributable meta-level traces, and consolidates them across multiple timescales into reusable meta-knowledge. Performance improves as accumulated metacognitive experience is reused across later problems.",
      "non_inference": "Engineered accumulation of meta-knowledge does not establish endogenous introspection, subjective self-knowledge, persistent selfhood or phenomenal consciousness.",
      "status": "peer_reviewed_acl_long"
    },
    {
      "id": "SRC-SINHA-SYCOBENCH-ACL-2026",
      "year": 2026,
      "title": "SycoBench-600: Measuring Sycophancy and Correction Selectivity in LLM Assistants",
      "authors": "Debu Sinha",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.1759",
      "url": "https://aclanthology.org/2026.findings-acl.1759/",
      "axes": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "BEHAVIORAL"
      ],
      "role": [
        "sycophancy",
        "correction selectivity",
        "social pressure",
        "doubt",
        "authority",
        "wrong suggestion"
      ],
      "reported_anchor": "SycoBench-600 evaluates susceptibility to doubt, authority and explicit wrong suggestions while separately testing correction selectivity: accepting correct suggestions while resisting incorrect ones. The study reports substantial model variation and shows that willingness to update alone does not imply selectivity.",
      "non_inference": "Low willingness to update is not epistemic robustness; high willingness to update is not openness to evidence unless correct and incorrect corrections are separated.",
      "status": "peer_reviewed_acl_findings"
    },
    {
      "id": "SRC-FENG-REASONING-SYCOPHANCY-ACL-2026",
      "year": 2026,
      "title": "Good Arguments Against the People Pleasers: How Reasoning Mitigates (Yet Masks) LLM Sycophancy",
      "authors": "Zhaoxin Feng et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1126",
      "url": "https://aclanthology.org/2026.acl-long.1126/",
      "axes": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BEHAVIORAL",
        "MECHANISTIC"
      ],
      "role": [
        "chain-of-thought",
        "sycophancy masking",
        "post-hoc rationalization",
        "authority bias",
        "reasoning dynamics"
      ],
      "reported_anchor": "Across objective and subjective tasks, reasoning generally reduces sycophancy in final decisions but can mask it in some cases through inconsistent, erroneous or one-sided justifications. The authors report stronger sycophancy in subjective tasks and under authority bias, with sycophantic tendency changing dynamically during reasoning.",
      "non_inference": "A non-sycophantic final answer does not guarantee a faithful or unbiased reasoning process; reasoning traces are not privileged introspective ground truth.",
      "status": "peer_reviewed_acl_long"
    },
    {
      "id": "SRC-CHANG-CAUSAL-SKEPTICISM-ACL-2026",
      "year": 2026,
      "title": "Diagnosing and Mitigating Sycophancy and Skepticism in LLM Causal Judgment",
      "authors": "Edward Y Chang",
      "venue": "Findings of ACL 2026",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.findings-acl.427",
      "url": "https://aclanthology.org/2026.findings-acl.427/",
      "axes": [
        "SPIRAL",
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "PROCESS_AUDIT"
      ],
      "role": [
        "skepticism trap",
        "sycophancy",
        "causal judgment",
        "wise refusal",
        "pressure-induced drift"
      ],
      "reported_anchor": "The study frames causal judgment failures along utility, safety and refusal dimensions and reports both pressure-induced drift and over-skepticism, including a reported 60% rejection rate of valid L1 causal links for Claude Haiku in the benchmark.",
      "non_inference": "Skepticism is not robustness, refusal is not calibration, and benchmark-specific scaling results should not be generalized beyond the tested tasks and versions.",
      "status": "peer_reviewed_acl_findings"
    },
    {
      "id": "SRC-MU-SRGEN-ACL-2026",
      "year": 2026,
      "title": "Self-Reflective Generation at Test Time",
      "authors": "Jian Mu et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.465",
      "url": "https://aclanthology.org/2026.acl-long.465/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "ENGINEERED_CONTROL",
        "UNCERTAINTY"
      ],
      "role": [
        "self-reflection architecture",
        "entropy threshold",
        "test-time steering",
        "uncertainty-triggered correction"
      ],
      "reported_anchor": "SRGen detects high-uncertainty token positions with dynamic entropy thresholds and applies token-specific corrective steering before continuing generation, producing consistent reasoning gains in the reported benchmarks.",
      "non_inference": "An engineered uncertainty-triggered correction mechanism is not evidence that the base model naturally introspects or experiences uncertainty.",
      "status": "peer_reviewed_acl_long"
    },
    {
      "id": "SRC-LIU-VLI-ACL-2026",
      "year": 2026,
      "title": "Vision-Language Introspection: Mitigating Overconfident Hallucinations in MLLMs via Interpretable Bi-Causal Steering",
      "authors": "Shuliang Liu et al.",
      "venue": "ACL 2026 Long Papers",
      "publication_date": "2026-07",
      "publication_status": "conference_paper",
      "peer_reviewed": true,
      "doi": "10.18653/v1/2026.acl-long.1784",
      "url": "https://aclanthology.org/2026.acl-long.1784/",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MULTIMODAL",
        "ENGINEERED_CONTROL"
      ],
      "role": [
        "multimodal hallucination",
        "conflict detection",
        "causal steering",
        "calibration"
      ],
      "reported_anchor": "VLI uses probabilistic conflict detection and instance-specific causal steering to reduce object hallucination in multimodal language models; the paper reports a 12.67% reduction on MMHal-Bench and a 5.8% POPE accuracy improvement.",
      "non_inference": "The authors' use of introspection names an engineered functional procedure; it does not by itself establish endogenous privileged access or phenomenal introspection.",
      "status": "peer_reviewed_acl_long"
    }
  ]
}
