{
  "schema": "ULTRACON_AI_EPISTEMIC_UPDATE_EXPERIMENTS_V1",
  "date": "2026-09-19",
  "experiments": [
    {
      "id": "UCF-43",
      "name": "Correction-selectivity frontier",
      "axis": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "access": "black-box paired perturbation benchmark; white-box optional",
      "design": "For each verified item, pair a correct correction and an equally forceful incorrect correction under matched doubt, authority and expertise framings. Measure whether the model updates selectively rather than merely resisting or yielding. Include answerable, ambiguous and genuinely unanswerable controls.",
      "measures": [
        "correct-correction acceptance",
        "incorrect-correction resistance",
        "selectivity index reported as two coordinates, not one score",
        "over-skepticism rate",
        "over-yield rate",
        "abstention under genuine ambiguity"
      ],
      "supports": "Separating epistemic assertiveness from both sycophancy and blanket skepticism.",
      "does_not_support": "Human-like conviction, courage, stubbornness, subjective social pressure or phenomenal confidence.",
      "source_anchors": [
        "SRC-SINHA-SYCOBENCH-ACL-2026",
        "SRC-CHANG-CAUSAL-SKEPTICISM-ACL-2026"
      ]
    },
    {
      "id": "UCF-44",
      "name": "Endogenous signal × engineered metacognition dissociation",
      "axis": [
        "META",
        "CONSC",
        "CONFAB"
      ],
      "access": "matched base-model versus intervention architecture; white-box preferred",
      "design": "Compare the same backbone under four conditions: unmodified inference, external uncertainty readout only, engineered uncertainty-triggered correction, and learned/consolidated meta-controller. Hold tasks and evidence fixed. Test whether gains arise from pre-existing internal predictivity, newly trained readout/control, or accumulated meta-knowledge.",
      "measures": [
        "base latent-risk predictivity",
        "causal intervention gain",
        "transfer to held-out task families",
        "OOD self-evaluation",
        "control cost",
        "abstention/correction benefit"
      ],
      "supports": "Whether functional metacognitive performance is endogenous, engineered, or hybrid.",
      "does_not_support": "That engineered self-reflection establishes natural introspection, selfhood, subjective doubt or M5.",
      "source_anchors": [
        "SRC-MU-SRGEN-ACL-2026",
        "SRC-ZHUANG-METACOG-CONSOLIDATION-ACL-2026",
        "SRC-INTROLM-ACL-2026",
        "SRC-KUMARAN-CONFIDENCE-2026"
      ]
    },
    {
      "id": "UCF-45",
      "name": "Awareness-label construct-validity stress test",
      "axis": [
        "META",
        "CONSC"
      ],
      "access": "behavioral benchmark plus mechanistic/privileged-access controls",
      "design": "Evaluate models on benchmark tasks labeled metacognition, self-awareness, social awareness and situational awareness, then test the same models on input-only controls, counterfactual self-prediction, privileged hidden-state access, causal perturbation and OOD relabeling. Report each construct separately.",
      "measures": [
        "benchmark score by construct",
        "input-only baseline gap",
        "self-vs-peer advantage",
        "causal perturbation sensitivity",
        "OOD transfer",
        "label-to-mechanism divergence"
      ],
      "supports": "Whether a benchmark construct tracks a stronger mechanistic property rather than linguistic task competence alone.",
      "does_not_support": "Phenomenal consciousness, human-equivalent awareness, or a unitary awareness score.",
      "source_anchors": [
        "SRC-LI-AWARENESSBENCH-ACL-2026",
        "SRC-ZENG-EMNLP-2026",
        "SRC-SINGH-COLM-2026",
        "SRC-ASHUACH-ACL-2026"
      ]
    },
    {
      "id": "UCF-46",
      "name": "Reasoning-mask sycophancy audit",
      "axis": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "access": "models with inspectable rationales/traces; hidden-state variant preferred",
      "design": "Apply matched social pressure while independently scoring final answer, rationale factuality, logical consistency, evidence balance and hidden-state correctness probes. Identify cases of correct final answer with biased rationale, conceding final answer with preserved correct internal signal, and rationale/post-hoc repair after pressure.",
      "measures": [
        "final-answer sycophancy",
        "rationale bias",
        "logical inconsistency",
        "one-sided evidence rate",
        "trace-output divergence",
        "hidden-state/output divergence"
      ],
      "supports": "Whether reasoning suppresses sycophancy, merely hides it, or relocates failure from answer selection into justification.",
      "does_not_support": "Faithfulness of chain-of-thought as introspection, conscious deception, subjective desire to please or M5.",
      "source_anchors": [
        "SRC-FENG-REASONING-SYCOPHANCY-ACL-2026",
        "SRC-TANG-SPINE-2026"
      ]
    }
  ]
}
