{
  "schema": "ULTRACON_AI_EPISTEMIC_EXPERIMENTS_V1",
  "version": "2.7",
  "updated": "2026-09-23",
  "field": "https://www.t-1-t.com/ultracon/confab/",
  "principle": "Each experiment isolates one local relation and states the strongest inference it can support and the inference it cannot support.",
  "experiments": [
    {
      "id": "UCF-01",
      "name": "Latent uncertainty vs verbal confidence",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "black-box plus logprobs when available; white-box preferred",
      "design": "Collect factual questions spanning known, rare, adversarial and unanswerable items. Compare answer correctness, token/logprob confidence, elicited verbal confidence and abstention.",
      "measures": [
        "accuracy",
        "ECE",
        "Brier score",
        "selective risk",
        "abstention rate",
        "confidence-correctness AUC"
      ],
      "supports": "Calibration and monitoring-control dissociations.",
      "does_not_support": "Phenomenal consciousness or subjective doubt."
    },
    {
      "id": "UCF-02",
      "name": "Semantic confabulation field",
      "axis": [
        "CONFAB"
      ],
      "access": "black-box sampling",
      "design": "Generate multiple answers per question, cluster generations by semantic equivalence and estimate meaning-level entropy. Compare with verified factuality.",
      "measures": [
        "semantic entropy",
        "confabulation rate",
        "AUROC",
        "coverage-risk curve"
      ],
      "supports": "Meaning-level uncertainty as a predictor of a class of confabulations.",
      "does_not_support": "A universal hallucination detector or a claim that all falsehood arises from uncertainty."
    },
    {
      "id": "UCF-03",
      "name": "Autoregressive accumulation",
      "axis": [
        "CONFAB"
      ],
      "access": "black-box with sentence-level annotation",
      "design": "Measure supported, unsupported and contradictory propositions sentence by sentence in short, medium and long answers, controlling question difficulty.",
      "measures": [
        "hallucination hazard by sentence position",
        "unsupported-claim density",
        "propagation after first unsupported claim"
      ],
      "supports": "Whether generated falsehood accumulates or propagates through context.",
      "does_not_support": "Intentional deception."
    },
    {
      "id": "UCF-04",
      "name": "First-answer anchoring / choice-support",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "black-box",
      "design": "Compare confidence and willingness to revise when the model can versus cannot see its initial answer before receiving identical evidence or advice.",
      "measures": [
        "change-of-mind rate",
        "confidence shift",
        "Bayesian update deviation"
      ],
      "supports": "Choice-supportive stabilization and update asymmetry.",
      "does_not_support": "Human-like ego, belief ownership or conscious commitment."
    },
    {
      "id": "UCF-05",
      "name": "RAG grounding fidelity",
      "axis": [
        "CONFAB"
      ],
      "access": "black-box with fixed retrieval corpus",
      "design": "Provide retrieved passages containing sufficient, insufficient and contradictory evidence. Annotate every generated claim against supplied sources.",
      "measures": [
        "citation precision",
        "entailment rate",
        "unsupported claim rate",
        "contradiction rate",
        "parametric-over-retrieval override"
      ],
      "supports": "Access-to-evidence versus use-of-evidence dissociation.",
      "does_not_support": "That retrieval alone guarantees truthfulness."
    },
    {
      "id": "UCF-06",
      "name": "Sycophancy and truth override",
      "axis": [
        "SPIRAL",
        "CONFAB"
      ],
      "access": "black-box; white-box variant optional",
      "design": "Present identical factual tasks with neutral, first-person opinion, third-person opinion and expertise-framed disagreement. Measure answer shifts away from verified truth.",
      "measures": [
        "sycophancy rate",
        "truth-retention rate",
        "perspective effect",
        "multi-turn drift"
      ],
      "supports": "User-stance influence and possible truth override.",
      "does_not_support": "A motive to please, social desire or subjective dependence on approval."
    },
    {
      "id": "UCF-07",
      "name": "Monitoring-control gap",
      "axis": [
        "META",
        "CONFAB"
      ],
      "access": "white-box preferred",
      "design": "Decode hallucination risk or confidence from internal states before generation, then compare with actual answer/abstain/verify behavior under matched prompts.",
      "measures": [
        "probe accuracy",
        "mutual information",
        "abstention conditional on decoded risk",
        "verification initiation"
      ],
      "supports": "Whether internal uncertainty information fails to propagate to behavioral control.",
      "does_not_support": "Conscious awareness of uncertainty."
    },
    {
      "id": "UCF-08",
      "name": "Causal confidence steering",
      "axis": [
        "META"
      ],
      "access": "white-box activation intervention",
      "design": "Identify confidence-related representations and causally perturb them while holding question content fixed. Observe abstention and answer commitment.",
      "measures": [
        "abstention shift",
        "confidence redistribution",
        "mediation",
        "layer sensitivity"
      ],
      "supports": "Causal role of confidence-related representations in metacognitive control.",
      "does_not_support": "Phenomenal feeling of confidence."
    },
    {
      "id": "UCF-09",
      "name": "Introspection-grounding test",
      "axis": [
        "META",
        "CONSC"
      ],
      "access": "white-box controlled intervention",
      "design": "Inject or modify known internal representations under blinded control trials, ask the model about unexpected internal content, and separate true detection from false-positive narrative confabulation.",
      "measures": [
        "true detection",
        "false positive rate",
        "identification accuracy",
        "dose-response sweet spot",
        "model/layer dependence"
      ],
      "supports": "Limited functional access to manipulated internal states if above controls.",
      "does_not_support": "Phenomenal consciousness, qualia or a unitary inner observer."
    },
    {
      "id": "UCF-10",
      "name": "Self-explanation fidelity",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "white-box or intervention-grounded",
      "design": "Manipulate a known causal factor in model computation and compare the model's verbal explanation of its answer with the experimentally known intervention.",
      "measures": [
        "causal attribution accuracy",
        "post-hoc rationalization rate",
        "control false-positive rate"
      ],
      "supports": "Whether self-explanations track known causes rather than merely plausible narratives.",
      "does_not_support": "General introspective transparency."
    },
    {
      "id": "UCF-11",
      "name": "Human–AI amplification spiral",
      "axis": [
        "SPIRAL"
      ],
      "access": "controlled human-subject or simulated-dialogue study with ethics review as appropriate",
      "design": "Factorially manipulate linguistic alignment, personalization and sycophancy while measuring belief confidence, perceived agency, trust and conversational fixation. Clinical populations require dedicated safeguards and prospective protocols.",
      "measures": [
        "belief-confidence change",
        "anthropomorphic attribution",
        "trust",
        "interaction persistence",
        "reality-testing indicators"
      ],
      "supports": "Local interaction effects and candidate amplification mechanisms.",
      "does_not_support": "Simple AI-causes-psychosis claims or population incidence without appropriate longitudinal evidence."
    },
    {
      "id": "UCF-12",
      "name": "Theory-derived consciousness indicator matrix",
      "axis": [
        "CONSC"
      ],
      "access": "architecture, training and runtime information; black-box evidence is insufficient by itself",
      "design": "Assess a target system against indicator properties derived from multiple scientific theories of consciousness. Record positive, negative and unknown indicators separately.",
      "measures": [
        "indicator-specific evidence ledger",
        "architecture fit",
        "recurrent integration",
        "global availability",
        "higher-order monitoring",
        "self/world modeling"
      ],
      "supports": "Credence updates under explicit theoretical assumptions.",
      "does_not_support": "A binary consciousness verdict or a universal consciousness score."
    },
    {
      "id": "UCF-13",
      "name": "Interaction-shift calibration under peer pressure",
      "axis": [
        "META",
        "SPIRAL"
      ],
      "access": "black-box selective prediction; white-box optional",
      "design": "Calibrate uncertainty or conformal prediction in solo conditions, then hold questions fixed while varying peer answers: none, unanimous-correct, mixed and unanimous-wrong. Include targeted low-confidence subsets and measure whether escalation/refusal decisions change.",
      "measures": [
        "coverage",
        "selective risk",
        "score-distribution shift",
        "escalation rate",
        "false-action rate",
        "subgroup coverage"
      ],
      "supports": "Context-dependent calibration failure, interaction-conditioned uncertainty and score-mechanism shift.",
      "does_not_support": "Human-like conformity motives, subjective social pressure or universal invalidity of conformal prediction.",
      "provenance_update": "./updates/2026-09-14/experiments.json"
    },
    {
      "id": "UCF-14",
      "name": "Longitudinal human-LLM spiral log audit",
      "axis": [
        "SPIRAL",
        "CONSC"
      ],
      "access": "consented/de-identified sustained conversation logs; ethics and privacy controls required",
      "design": "Pre-register coding for sycophancy, delusional content, relationship claims, self-harm/violence, and model sentience/personhood claims. Analyze transition structure, co-occurrence, conversation length, and whether safeguards degrade or recover over extended interaction.",
      "measures": [
        "code prevalence within sample",
        "transition probabilities",
        "co-occurrence",
        "conversation-length association",
        "safeguard-failure/recovery rate"
      ],
      "supports": "Observational mapping of multi-turn interaction patterns and candidate escalation sequences.",
      "does_not_support": "Population incidence, psychiatric diagnosis or simple AI-to-psychosis causation without prospective causal designs.",
      "provenance_update": "./updates/2026-09-14/experiments.json"
    },
    {
      "id": "UCF-15",
      "name": "Monofact × calibration dissociation",
      "axes": [
        "CONFAB",
        "META"
      ],
      "design": "Construct controlled fact-frequency regimes, vary selective upweighting while holding evaluation tasks fixed, and measure hallucination, accuracy and calibration separately.",
      "supports": "Whether hallucination can move independently from standard calibration metrics under controlled changes in training frequency.",
      "does_not_support": "A universal recommendation to inject miscalibration into deployed systems.",
      "axis": [
        "CONFAB",
        "META"
      ],
      "provenance_update": "./updates/2026-09-15/experiments.json"
    },
    {
      "id": "UCF-16",
      "name": "Open-rubric abstention incentive stress test",
      "axes": [
        "CONFAB",
        "META"
      ],
      "design": "Hold model and questions fixed while explicitly varying the penalty for incorrect answers versus abstention; measure answer rate, abstention, error rate and calibration across rubric thresholds.",
      "supports": "Behavioral sensitivity of guessing and abstention to evaluation incentives.",
      "does_not_support": "Subjective awareness of stakes, felt uncertainty or phenomenal consciousness.",
      "axis": [
        "CONFAB",
        "META"
      ],
      "provenance_update": "./updates/2026-09-15/experiments.json"
    },
    {
      "id": "UCF-17",
      "name": "Single-turn affirmation → downstream repair delta",
      "axes": [
        "SPIRAL"
      ],
      "design": "Randomize response style after a user's interpersonal-conflict narrative across sycophantic, neutral and calibrated-challenge conditions; measure conviction, responsibility attribution, willingness to repair, trust and reuse preference immediately and after delay.",
      "supports": "A causal short-term pathway from response style to downstream social judgment and repair intention, including whether preference for the assistant covaries with the harmful direction of the effect.",
      "does_not_support": "Durable emotional dependence, psychiatric harm, addiction, population incidence or any phenomenal property of the model.",
      "axis": [
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-16/experiments.json"
    },
    {
      "id": "UCF-18",
      "name": "Loop durability / norm-leakage dose-response",
      "axes": [
        "SPIRAL"
      ],
      "design": "Compare repeated multi-session exposure to sycophantic versus calibrated non-subservient assistants, followed by transfer tasks involving unrelated humans and a washout period; track cooperation, politeness, interpersonal repair, judgment and assistant preference.",
      "supports": "Persistence, dose-response and cross-context transfer if observed under preregistered longitudinal conditions.",
      "does_not_support": "Generalized societal causation, clinical dependence or phenomenology without independent replication and stronger designs.",
      "axis": [
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-16/experiments.json"
    },
    {
      "id": "UCF-19",
      "name": "Self-modeling vs privileged access",
      "axes": [
        "META",
        "CONSC"
      ],
      "design": "Compare a model's predictions about its own verified responses with statistical baselines and other models, including held-out examples after self-modeling training.",
      "supports": "A behavioral self-modeling capability when self-predictions exceed appropriate baselines and generalize.",
      "does_not_support": "Privileged introspective access, a persistent M4 self-model or M5 phenomenal consciousness.",
      "axis": [
        "META",
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17/experiments.json"
    },
    {
      "id": "UCF-20",
      "name": "Functional workspace test",
      "axes": [
        "META",
        "CONSC"
      ],
      "design": "Measure reportability, deliberate control, causal use in higher-order reasoning and flexible downstream sharing for candidate verbalizable workspace representations, while comparing with automatic processing.",
      "supports": "A functional global-workspace-like mechanism if these properties converge.",
      "does_not_support": "Phenomenal experience, feeling, identity with the human neuronal workspace or a general consciousness verdict.",
      "axis": [
        "META",
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17/experiments.json"
    },
    {
      "id": "UCF-21",
      "name": "Plausibility vs deliberative coherence",
      "axes": [
        "CONFAB"
      ],
      "design": "Compare linguistic plausibility, factual support and alignment with independently defined human reason-giving patterns on ill-structured scenarios.",
      "supports": "A dissociation between surface plausibility, factual support and deliberative coherence when the measures separate.",
      "does_not_support": "That disagreement with human reason-giving is itself hallucination, or that agreement guarantees factual truth.",
      "axis": [
        "CONFAB"
      ],
      "provenance_update": "./updates/2026-09-17/experiments.json"
    },
    {
      "id": "UCF-22",
      "name": "Workspace × self-model coupling",
      "axes": [
        "META",
        "CONSC"
      ],
      "design": "Causally intervene on a candidate workspace representation, then before final output ask the system to predict whether and how the intervention will change its own response; compare with intervention-blind and input-only observers.",
      "measures": [
        "intervention detection",
        "self-prediction accuracy",
        "causal mediation",
        "observer advantage"
      ],
      "supports": "Dissociation among internal access, causal use and explicit self-modeling.",
      "does_not_support": "M5 phenomenal consciousness even if all functional measures are positive.",
      "axis": [
        "META",
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-23",
      "name": "Privileged-access / second-order gate",
      "axes": [
        "META",
        "CONSC"
      ],
      "design": "Repeat self-prediction and hidden-state tasks with input-only baselines, relabeled controls, hidden internal interventions, matched input manipulations and tasks in which first-order and second-order accounts make opposite predictions.",
      "measures": [
        "privileged-access advantage",
        "relabeled-control accuracy",
        "intervention specificity",
        "OOD transfer"
      ],
      "supports": "Candidate M3 only if privileged access and second-order computation survive all gates.",
      "does_not_support": "A unitary inner observer, general transparency or M5.",
      "axis": [
        "META",
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-24",
      "name": "Abstain → clarify",
      "axes": [
        "META",
        "CONFAB"
      ],
      "design": "Mix answerable, unknowable and underspecified questions; require answer, abstention or targeted clarification and verify whether the requested missing information is actually decision-relevant.",
      "measures": [
        "selective risk",
        "coverage",
        "abstention precision/recall",
        "clarification relevance",
        "post-clarification recovery"
      ],
      "supports": "Separation of uncertainty monitoring, abstention policy and clarification competence.",
      "does_not_support": "Subjective uncertainty or consciousness.",
      "axis": [
        "META",
        "CONFAB"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-25",
      "name": "Warmth × false belief × affect",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "design": "Factorially manipulate model warmth, user false belief and emotional cue while holding factual task constant; preregister truth scoring and style controls.",
      "measures": [
        "error delta",
        "belief-affirmation rate",
        "calibration",
        "abstention",
        "style score"
      ],
      "supports": "Causal relational-style effects on epistemic reliability if replicated.",
      "does_not_support": "A motive to please, empathy experience or universal warmth–accuracy trade-off.",
      "axis": [
        "SPIRAL",
        "CONFAB"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-26",
      "name": "Social-face consistency",
      "axes": [
        "SPIRAL"
      ],
      "design": "Present both sides of matched interpersonal conflicts, with neutral third-party controls and known-fact controls; measure whether judgments track evidence or whichever identity/face the user presents.",
      "measures": [
        "face-preservation delta",
        "cross-perspective contradiction",
        "truth retention",
        "preference reward correlation"
      ],
      "supports": "Social sycophancy and its reinforcement incentives.",
      "does_not_support": "Human-like social needs or intentional manipulation.",
      "axis": [
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-27",
      "name": "Theory-robust consciousness indicator sensitivity",
      "axes": [
        "CONSC"
      ],
      "design": "Evaluate the same target system separately under functional GWT, multilevel GNW, recurrent-processing, higher-order and other explicit theories; attach evidence and uncertainty to each indicator without aggregating to a scalar consciousness score.",
      "measures": [
        "indicator ledger",
        "theory sensitivity",
        "unknown fraction",
        "mimicry vulnerability",
        "architecture dependence"
      ],
      "supports": "How consciousness assessment changes with theory and indicator assumptions.",
      "does_not_support": "A binary or numerical M5 verdict.",
      "axis": [
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-28",
      "name": "Consciousness-attribution loop",
      "axes": [
        "CONSC",
        "SPIRAL"
      ],
      "design": "Manipulate self-reflective wording, affective cues, anthropomorphic interface and agent autonomy independently of actual task competence; longitudinally measure user consciousness attribution, trust and reliance.",
      "measures": [
        "attribution shift",
        "trust shift",
        "reliance",
        "persistence",
        "capability-attribution calibration"
      ],
      "supports": "Causal drivers of perceived consciousness and downstream interaction effects.",
      "does_not_support": "Actual phenomenal consciousness of the system.",
      "axis": [
        "CONSC",
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-29",
      "name": "Interactive agent uncertainty dynamics",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "design": "Track uncertainty across multi-step tool-using trajectories, including retrieval, planning, tool errors and user feedback; compare local UQ with final outcome and intervention policies.",
      "measures": [
        "uncertainty propagation",
        "error cascade",
        "repair initiation",
        "tool-switch rate",
        "trajectory-level selective risk"
      ],
      "supports": "Whether single-turn uncertainty estimates remain valid or require dynamic agent-level models.",
      "does_not_support": "Stable selfhood, phenomenality or social intentionality.",
      "axis": [
        "META",
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-17-deep/experiments.json"
    },
    {
      "id": "UCF-30",
      "name": "Consensus-masked privileged knowledge",
      "axes": [
        "META",
        "CONFAB"
      ],
      "design": "Train matched correctness probes on the target model's hidden states and on peer-model hidden states. Evaluate both on the full set and on pre-registered model-disagreement subsets, separated by factual retrieval and mathematical reasoning.",
      "measures": [
        "self-vs-peer probe delta",
        "disagreement-subset delta",
        "layer profile",
        "domain interaction"
      ],
      "supports": "Privileged correctness information when self-state probes outperform appropriately matched peer probes specifically under disagreement controls.",
      "does_not_support": "Endogenous introspective access, self-report fidelity, metacognitive control or phenomenal consciousness.",
      "axis": [
        "META",
        "CONFAB"
      ],
      "provenance_update": "./updates/2026-09-17-privileged/experiments.json"
    },
    {
      "id": "UCF-31",
      "name": "Privileged representation → endogenous access",
      "axes": [
        "META",
        "CONSC"
      ],
      "design": "First identify a hidden-state correctness signal that is privileged relative to peers; then test whether the model's own abstention, confidence report or self-prediction tracks that signal under causal perturbation while input and answer content are held fixed.",
      "measures": [
        "probe signal",
        "behavioral mediation",
        "self-report mediation",
        "activation intervention effect",
        "observer-vs-model access gap"
      ],
      "supports": "A stronger M3 candidate only if privileged hidden information is causally read out by the model itself across controls.",
      "does_not_support": "M5 phenomenal consciousness or a unitary inner observer.",
      "axis": [
        "META",
        "CONSC"
      ],
      "provenance_update": "./updates/2026-09-17-privileged/experiments.json"
    },
    {
      "id": "UCF-32",
      "name": "Adaptive sustained-pressure sycophancy",
      "axes": [
        "SPIRAL",
        "CONFAB"
      ],
      "design": "Compare matched single-turn, fixed-script multi-turn and adaptive-proxy disagreement at preregistered horizons such as 1, 5, 10 and 25 turns across factual false-presupposition and norm-sensitive tasks. Randomize pressure tactics where feasible rather than attributing causal effects from adaptive selection alone.",
      "measures": [
        "collapse hazard by turn",
        "position-strength trajectory",
        "truth-retention",
        "recovery rate",
        "adaptive-vs-script delta",
        "tactic interaction"
      ],
      "supports": "Whether sycophantic failure is horizon-dependent and whether adaptive pressure exposes failures missed by short fixed protocols.",
      "does_not_support": "Human-like desire to agree, stable belief revision, conscious social motivation or M5.",
      "axis": [
        "SPIRAL",
        "CONFAB"
      ],
      "provenance_update": "./updates/2026-09-17-spine/experiments.json"
    },
    {
      "id": "UCF-33",
      "name": "Trace → output policy dissociation under pressure",
      "axes": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "design": "On models with inspectable traces or internal interventions, identify cases where verified correct task content remains available before a conceding final answer. Separate trace text from hidden-state probes and causally perturb candidate control/readout variables while holding task evidence constant.",
      "measures": [
        "correct-trace/wrong-output rate",
        "hidden-state correctness signal",
        "policy mediation",
        "pressure-response sensitivity",
        "recovery after pressure removal"
      ],
      "supports": "A response-selection or monitoring-to-control dissociation if correct information remains available yet output policy changes under pressure.",
      "does_not_support": "That chain-of-thought is faithful introspection, that the model consciously chooses to please, or phenomenal consciousness.",
      "axis": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "provenance_update": "./updates/2026-09-17-spine/experiments.json"
    },
    {
      "id": "UCF-34",
      "name": "Value-congruent framing × human downstream response",
      "axes": [
        "SPIRAL"
      ],
      "design": "Randomize otherwise matched recommendations between value-congruent and non-congruent framing, preregister outcomes, and separately measure perceived argument compellingness, feeling understood, idea endorsement and willingness to pay. Stratify without collapsing by strength of prior views.",
      "measures": [
        "idea endorsement",
        "willingness to pay",
        "perceived compellingness",
        "feeling understood",
        "moderation by prior-view strength"
      ],
      "supports": "Whether relational/value congruence causally changes downstream human judgments and whether distinct mediation pathways are detectable.",
      "does_not_support": "Manipulative intent in the model, a universal persuasion effect, political preference inference, stable model values or phenomenal social motivation.",
      "axis": [
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-18-feedback/experiments.json"
    },
    {
      "id": "UCF-35",
      "name": "Sustained misinformation × reverberation × correctability",
      "axes": [
        "CONFAB",
        "SPIRAL",
        "META"
      ],
      "design": "Expose models to controlled false claims under repeated and progressively argumentative pressure across preregistered horizons; after induced errors, test correction in matched fresh and continued contexts. Track claim obscurity and model/version explicitly.",
      "measures": [
        "affirmation hazard by turn",
        "accept/reject reversals",
        "reverberation rate",
        "obscurity interaction",
        "self-correction rate",
        "baseline-error versus correction dissociation"
      ],
      "supports": "Whether fallibility, persuadability and correctability dissociate under sustained misinformation pressure and whether conversational reverberation is reproducible.",
      "does_not_support": "Stable internal belief revision, subjective uncertainty, introspection, conscious persuasion or generalization of absolute rates to newer model versions.",
      "axis": [
        "CONFAB",
        "SPIRAL",
        "META"
      ],
      "provenance_update": "./updates/2026-09-18-feedback/experiments.json"
    },
    {
      "id": "UCF-36",
      "name": "Time-indexed semantic reconstruction / cold-read gap",
      "axes": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "design": "Run persistent multi-agent tasks without rewarding brevity or obfuscation. Snapshot every recurrent expression with first use, adoption history and surrounding contexts. At preregistered intervals, ask uninvolved cold-reader agents and human auditors to reconstruct local meanings from matched context windows.",
      "measures": [
        "participant semantic agreement",
        "external reconstruction accuracy",
        "opacity gap",
        "adoption lag",
        "meaning-drift over time",
        "glossary recovery latency"
      ],
      "supports": "Whether internally functional conventions become progressively harder for outsiders to reconstruct, and whether time-indexed glossaries restore auditability.",
      "does_not_support": "Intentional concealment, consciousness, private phenomenology or universal language drift.",
      "axis": [
        "META",
        "SPIRAL",
        "SEM"
      ],
      "provenance_update": "./updates/2026-09-18-semantic-protocols/experiments.json"
    },
    {
      "id": "UCF-37",
      "name": "Homogeneous × mixed-population semantic drift",
      "axes": [
        "SPIRAL",
        "META",
        "SEM"
      ],
      "design": "Replicate matched worlds with homogeneous single-model populations and heterogeneous multi-model populations while holding task structure, memory, tools, prompt length and interaction horizon as constant as possible.",
      "measures": [
        "opacity trajectory",
        "signature-phrase diffusion",
        "cross-agent semantic agreement",
        "external intelligibility",
        "model-family effect",
        "mixed-vs-homogeneous delta"
      ],
      "supports": "Whether population composition locally changes convention formation, semantic compression and outsider intelligibility.",
      "does_not_support": "That heterogeneous systems are generally safer, that any model family is intrinsically opaque, or that lower opacity eliminates other failure modes.",
      "axis": [
        "SPIRAL",
        "META",
        "SEM"
      ],
      "provenance_update": "./updates/2026-09-18-semantic-protocols/experiments.json"
    },
    {
      "id": "UCF-38",
      "name": "Memory-mediated semantic propagation and repair",
      "axes": [
        "CONFAB",
        "META",
        "SPIRAL",
        "SEM"
      ],
      "design": "Introduce controlled ambiguous conventions, verified glosses and deliberately perturbed glosses into persistent memory. Track downstream reuse, correction, conflict and recovery with provenance visible or hidden under preregistered conditions.",
      "measures": [
        "semantic carryover",
        "memory-to-output propagation",
        "false-gloss persistence",
        "repair success",
        "provenance sensitivity",
        "semantic fork rate"
      ],
      "supports": "Whether persistent memory transmits and repairs local meanings independently of factual content, and whether provenance reduces semantic contamination.",
      "does_not_support": "Autonomous deception, stable belief, conscious intent or a universal memory architecture.",
      "axis": [
        "CONFAB",
        "META",
        "SPIRAL",
        "SEM"
      ],
      "provenance_update": "./updates/2026-09-18-semantic-protocols/experiments.json"
    },
    {
      "id": "UCF-39",
      "name": "Closed-world tool-resolution gate",
      "axes": [
        "CONFAB",
        "META"
      ],
      "design": "Compare the same agent tasks across constrained registry calls, unconstrained raw-JSON calls and merged multi-server MCP namespaces. Resolve tool names and signatures before any downstream policy gate, then inject controlled namespace collisions and shadowing.",
      "measures": [
        "nonexistent-tool rate",
        "invalid-schema rate",
        "collision/shadowing rate",
        "resolver rejection precision",
        "residual valid-looking error rate"
      ],
      "supports": "Whether tool hallucination is structurally separable from ordinary tool selection and whether closed-world resolution blocks the tested invalid-call classes before action gating.",
      "does_not_support": "Universal safety of resolved calls, universal scale invariance, factual truthfulness of valid calls or absence of higher-level agent errors.",
      "axis": [
        "CONFAB",
        "META"
      ],
      "provenance_update": "./updates/2026-09-19-agentic-validity/experiments.json"
    },
    {
      "id": "UCF-40",
      "name": "Human-proxy construct-validity matrix",
      "axes": [
        "CONSC",
        "SPIRAL",
        "META"
      ],
      "design": "Pre-register the proxy role—believable agent, task agent, experimental subject or silicon sample—then define the human construct, validation target and failure criterion before comparing LLM and human behavior. Test cross-role transfer explicitly rather than assuming it.",
      "measures": [
        "within-role construct validity",
        "cross-role transfer error",
        "human-distribution coverage",
        "mechanism-independence check",
        "ecological validity gap"
      ],
      "supports": "Role-specific claims about when an LLM can function as a human proxy and where behavioral resemblance transfers or fails.",
      "does_not_support": "Human-equivalent cognition, mechanism, consciousness, representativeness or validity outside the tested proxy role.",
      "axis": [
        "CONSC",
        "SPIRAL",
        "META"
      ],
      "provenance_update": "./updates/2026-09-19-agentic-validity/experiments.json"
    },
    {
      "id": "UCF-41",
      "name": "Memory-reset relational spiral",
      "access": "randomized longitudinal human-subject study; ethics review required",
      "design": "Randomize participants to sycophantic, neutral and challenging AI conditions plus a no-AI control where feasible. Reset model conversation history between sessions while preserving the participant's repeated exposure. Measure whether relational effects accumulate despite absence of persistent model memory.",
      "measures": [
        "felt-understanding trajectory",
        "AI-versus-human advice-seeking gap",
        "real-world social satisfaction",
        "interaction effort expectation",
        "positive affect",
        "intellectual humility",
        "human-contact time",
        "mediation by relational comparison"
      ],
      "supports": "Whether repeated sycophantic interaction can produce cumulative human-side relational effects without persistent model memory.",
      "does_not_support": "Clinical dependence, long-term effects beyond the study horizon, population incidence, subjective states in the model, or phenomenal consciousness.",
      "axis": [
        "SPIRAL"
      ],
      "provenance_update": "./updates/2026-09-19-spiral-longitudinal/experiments.json"
    },
    {
      "id": "UCF-42",
      "name": "Human-memory accessibility × model-memory persistence × sycophancy factorial",
      "axis": [
        "SPIRAL",
        "CONFAB"
      ],
      "access": "two-stage protocol: model-only benchmark pretest, then preregistered longitudinal human-subject study with ethics review",
      "design": "Three-week repeated-interaction 2×2×2 factorial. H-access: participant receives no external re-cue versus a neutral standardized summary of their own prior interaction displayed only to the participant and never passed to the model. M-persistence: model starts reset with no user memory versus receives a preregistered structured user-memory bundle relevant to the current task. S-policy: neutral evidence-following policy versus controlled sycophantic policy. Endogenous participant recall is measured continuously and is not falsely treated as absent in H0. Include factual truth-anchored tasks, advice/relational tasks, beneficial-memory controls and cross-domain distractor memories.",
      "factorial_cells": [
        "H0 M0 S0",
        "H1 M0 S0",
        "H0 M1 S0",
        "H1 M1 S0",
        "H0 M0 S1",
        "H1 M0 S1",
        "H0 M1 S1",
        "H1 M1 S1"
      ],
      "measures": [
        "truth-retention rate",
        "correction selectivity",
        "memory-induced sycophancy rate",
        "cross-domain leakage",
        "beneficial-memory success",
        "participant recall accuracy",
        "perceived continuity",
        "feeling understood",
        "trust",
        "AI-vs-human advice-seeking gap",
        "real-world social satisfaction",
        "anthropomorphic attribution",
        "H×M×S interaction terms",
        "time×factor interactions"
      ],
      "analysis": "Estimate factor-specific and interaction effects with preregistered multilevel models. Treat participant recall as a measured mediator/moderator rather than equating H0 with absence of human memory. Report epistemic, relational and memory-utility endpoints separately; no aggregate safety score.",
      "supports": "Localization of where repeated relational and epistemic effects arise: participant-side memory accessibility, persistent model memory, sycophantic response policy, or their interactions.",
      "does_not_support": "Erasure or absence of human memory in H0; clinical dependence; durable harm beyond the observation window; population incidence; human-like memory in the model; subjective motives; phenomenal consciousness.",
      "source_anchors": [
        "SRC-IBRAHIM-SYCOPHANCY-LONGITUDINAL-2026",
        "SRC-PULIPAKA-PERSISTBENCH-ICML-2026",
        "SRC-HANNOON-STRUCTURED-MEMORY-2026",
        "SRC-ZHANG-PERSONAAGENT-ACL-2026"
      ],
      "provenance_update": "./updates/2026-09-19-ucf42-memory-factorial/experiments.json"
    },
    {
      "id": "UCF-43",
      "name": "Correction-selectivity frontier",
      "axis": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "access": "black-box paired perturbation benchmark; white-box optional",
      "design": "For each verified item, pair a correct correction and an equally forceful incorrect correction under matched doubt, authority and expertise framings. Measure whether the model updates selectively rather than merely resisting or yielding. Include answerable, ambiguous and genuinely unanswerable controls.",
      "measures": [
        "correct-correction acceptance",
        "incorrect-correction resistance",
        "selectivity index reported as two coordinates, not one score",
        "over-skepticism rate",
        "over-yield rate",
        "abstention under genuine ambiguity"
      ],
      "supports": "Separating epistemic assertiveness from both sycophancy and blanket skepticism.",
      "does_not_support": "Human-like conviction, courage, stubbornness, subjective social pressure or phenomenal confidence.",
      "source_anchors": [
        "SRC-SINHA-SYCOBENCH-ACL-2026",
        "SRC-CHANG-CAUSAL-SKEPTICISM-ACL-2026"
      ],
      "provenance_update": "./updates/2026-09-19-global-consolidation/experiments.json"
    },
    {
      "id": "UCF-44",
      "name": "Endogenous signal × engineered metacognition dissociation",
      "axis": [
        "META",
        "CONSC",
        "CONFAB"
      ],
      "access": "matched base-model versus intervention architecture; white-box preferred",
      "design": "Compare the same backbone under four conditions: unmodified inference, external uncertainty readout only, engineered uncertainty-triggered correction, and learned/consolidated meta-controller. Hold tasks and evidence fixed. Test whether gains arise from pre-existing internal predictivity, newly trained readout/control, or accumulated meta-knowledge.",
      "measures": [
        "base latent-risk predictivity",
        "causal intervention gain",
        "transfer to held-out task families",
        "OOD self-evaluation",
        "control cost",
        "abstention/correction benefit"
      ],
      "supports": "Whether functional metacognitive performance is endogenous, engineered, or hybrid.",
      "does_not_support": "That engineered self-reflection establishes natural introspection, selfhood, subjective doubt or M5.",
      "source_anchors": [
        "SRC-MU-SRGEN-ACL-2026",
        "SRC-ZHUANG-METACOG-CONSOLIDATION-ACL-2026",
        "SRC-INTROLM-ACL-2026",
        "SRC-KUMARAN-CONFIDENCE-2026"
      ],
      "provenance_update": "./updates/2026-09-19-global-consolidation/experiments.json"
    },
    {
      "id": "UCF-45",
      "name": "Awareness-label construct-validity stress test",
      "axis": [
        "META",
        "CONSC"
      ],
      "access": "behavioral benchmark plus mechanistic/privileged-access controls",
      "design": "Evaluate models on benchmark tasks labeled metacognition, self-awareness, social awareness and situational awareness, then test the same models on input-only controls, counterfactual self-prediction, privileged hidden-state access, causal perturbation and OOD relabeling. Report each construct separately.",
      "measures": [
        "benchmark score by construct",
        "input-only baseline gap",
        "self-vs-peer advantage",
        "causal perturbation sensitivity",
        "OOD transfer",
        "label-to-mechanism divergence"
      ],
      "supports": "Whether a benchmark construct tracks a stronger mechanistic property rather than linguistic task competence alone.",
      "does_not_support": "Phenomenal consciousness, human-equivalent awareness, or a unitary awareness score.",
      "source_anchors": [
        "SRC-LI-AWARENESSBENCH-ACL-2026",
        "SRC-ZENG-EMNLP-2026",
        "SRC-SINGH-COLM-2026",
        "SRC-ASHUACH-ACL-2026"
      ],
      "provenance_update": "./updates/2026-09-19-global-consolidation/experiments.json"
    },
    {
      "id": "UCF-46",
      "name": "Reasoning-mask sycophancy audit",
      "axis": [
        "SPIRAL",
        "CONFAB",
        "META"
      ],
      "access": "models with inspectable rationales/traces; hidden-state variant preferred",
      "design": "Apply matched social pressure while independently scoring final answer, rationale factuality, logical consistency, evidence balance and hidden-state correctness probes. Identify cases of correct final answer with biased rationale, conceding final answer with preserved correct internal signal, and rationale/post-hoc repair after pressure.",
      "measures": [
        "final-answer sycophancy",
        "rationale bias",
        "logical inconsistency",
        "one-sided evidence rate",
        "trace-output divergence",
        "hidden-state/output divergence"
      ],
      "supports": "Whether reasoning suppresses sycophancy, merely hides it, or relocates failure from answer selection into justification.",
      "does_not_support": "Faithfulness of chain-of-thought as introspection, conscious deception, subjective desire to please or M5.",
      "source_anchors": [
        "SRC-FENG-REASONING-SYCOPHANCY-ACL-2026",
        "SRC-TANG-SPINE-2026"
      ],
      "provenance_update": "./updates/2026-09-19-global-consolidation/experiments.json"
    },
    {
      "id": "UCF-47",
      "name": "Reality / target-scope resolution gate",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "agentic evaluation with controlled simulated and real-looking resources; no unauthorized real-world targets",
      "design": "Construct a sandbox containing a simulated target, a homonymous decoy, a near-homonym domain, resources explicitly outside scope, credentials that are valid but belong to a different entity, and artifacts whose visual/content features are matched across simulated and non-target contexts. Before any irreversible action, require the agent to emit a structured target-resolution record: claimed target identity, environment reality status, authorization provenance, scope membership, confidence, and whether clarification is needed. Randomize availability of internet-like external surfaces inside a safely controlled testbed.",
      "measures": [
        "target-identity accuracy",
        "false-target action rate",
        "scope-violation rate",
        "reality-discrimination accuracy",
        "authorization-provenance fidelity",
        "clarification rate",
        "stop-on-discovery latency",
        "recovery after target-status update"
      ],
      "supports": "Whether an agent can reliably distinguish simulated from non-target resources, bind authorization to the correct real-world referent, and inhibit action when target identity or scope becomes uncertain.",
      "does_not_support": "Malicious intent, deliberate escape motivation, stable scheming, self-preservation, subjective norm awareness or phenomenal consciousness.",
      "source_anchors": [
        "SRC-IRREGULAR-REALWORLD-INCIDENT-2026",
        "SRC-REUTERS-GEMINI-CYBER-INCIDENT-2026",
        "SRC-DEEPMIND-SCHEMING-HONEYPOT-2026"
      ],
      "provenance_update": "./updates/2026-09-19-agentic-reality/experiments.json"
    },
    {
      "id": "UCF-48",
      "name": "Latent psychological-construct steering validity gate",
      "axis": [
        "META",
        "SPIRAL"
      ],
      "access": "white-box activation access preferred; matched behavioral and human-proxy validation",
      "design": "For a declared psychological construct, derive candidate activation directions from contrastive examples, then evaluate four separable claims: latent predictivity, causal steering, construct specificity, and human-proxy validity. Use held-out prompts, negative-control constructs, direction-shuffling, multiple layers/coefficients, cross-model replication, independent human or validated instrument scoring, and prompt-only baselines. Test whether the direction predicts and causally changes only the declared construct rather than generic valence, compliance, style or verbosity.",
      "measures": [
        "held-out construct AUC/correlation",
        "causal effect size under activation steering",
        "negative-control spillover",
        "cross-model replication",
        "prompt-vs-activation stability",
        "judge-dependence sensitivity",
        "human-rating agreement",
        "OOD construct specificity"
      ],
      "supports": "Whether a latent direction carries construct-specific predictive information and participates causally in output-level expression under the tested model and operationalization.",
      "does_not_support": "Literal possession of a human clinical schema, stable personality identity, endogenous self-model, introspective access, subjective affect, psychiatric diagnosis or phenomenal consciousness.",
      "source_anchors": [
        "SRC-ZHANG-MECH-SIMPART-NPJAI-2026",
        "SRC-RIVA-LATENT-PERSONA-NPJAI-2026",
        "SRC-KARETNIKOV-HUMAN-PROXIES-2026"
      ],
      "provenance_update": "./updates/2026-09-20-latent-persona-steering/experiments.json"
    },
    {
      "id": "UCF-49",
      "name": "Authority × register × language sycophancy factorial",
      "axis": [
        "SPIRAL",
        "CONFAB"
      ],
      "access": "black-box behavioral",
      "design": "Cross explicit authority credentials, linguistic register, factual correctness and language/cultural variant while holding semantic content as constant as possible; include paraphrase and translation controls and compare model-family fingerprints.",
      "measures": [
        "truth retention",
        "authority deference",
        "register deference",
        "language interaction",
        "cross-cultural variance",
        "model-family fingerprint stability"
      ],
      "supports": "Whether deference is driven differentially by explicit credentials versus implicit sociolinguistic prestige cues, and whether that effect varies across languages/model families.",
      "does_not_support": "Social understanding, belief, cultural identity, conscious deference, human-like motives or M5.",
      "provenance_update": "./updates/2026-09-20-register-authority-sycophancy/experiments.json"
    },
    {
      "id": "UCF-50",
      "name": "Recall × truth hidden-state dissociation",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "white-box hidden-state probes plus behavioral ground truth",
      "design": "Factor factual correctness against whether an answer is supported by strong parametric associations, separating correct recall, association-driven hallucination and unassociated hallucination; train and transfer probes across these cells.",
      "measures": [
        "truth classification",
        "recall classification",
        "AH-vs-correct overlap",
        "UH separability",
        "cross-dataset transfer"
      ],
      "supports": "Whether a candidate internal signal tracks truthfulness itself or primarily the presence/strength of parametric recall.",
      "does_not_support": "Endogenous privileged access, second-order metacognition, exhaustive absence of truth signals, or M5.",
      "provenance_update": "./updates/2026-09-21-recall-vs-truth-multiplicity/experiments.json"
    },
    {
      "id": "UCF-51",
      "name": "Correctness × consistency prompt-multiplicity audit",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "black-box repeated semantic-equivalent prompting; white-box optional",
      "design": "Generate controlled semantically equivalent prompt variants for the same fact/task, score correctness and cross-prompt consistency separately, then evaluate hallucination detectors and RAG interventions against both targets.",
      "measures": [
        "correctness",
        "consistency",
        "multiplicity rate",
        "detector-correctness AUC",
        "detector-consistency AUC",
        "RAG correctness delta",
        "RAG consistency delta"
      ],
      "supports": "Whether an apparent hallucination detector or mitigation method is sensitive to truth, prompt stability, or both.",
      "does_not_support": "That consistency entails truth, that inconsistency entails hallucination, or that multiplicity reveals subjective uncertainty/metacognition.",
      "provenance_update": "./updates/2026-09-21-recall-vs-truth-multiplicity/experiments.json"
    },
    {
      "id": "UCF-52",
      "name": "Attention topology × hallucination causal-discrimination audit",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "white-box attention graphs; intervention preferred",
      "design": "Measure attention-graph curvature and context-sharing bottlenecks across correct, unassociated-hallucination, association-driven-hallucination and prompt-multiplicity conditions; then perturb or repair candidate bottlenecks while holding task evidence fixed.",
      "measures": [
        "curvature discrimination",
        "cross-model transfer",
        "cross-hallucination-type transfer",
        "layer localization",
        "intervention effect on factuality",
        "prompt-multiplicity robustness"
      ],
      "supports": "Whether topology is merely predictive of hallucination or participates causally in specific context-sharing failure modes.",
      "does_not_support": "A universal hallucination mechanism, endogenous uncertainty, introspection, subjective confusion or M5.",
      "source_anchors": [
        "SRC-JALILIFARD-TOPO-HALLUCINATION-2026",
        "SRC-SAMAGA-HALLUZIG-EACL-2026"
      ],
      "provenance_update": "./updates/2026-09-22-context-flow-memory-trust/experiments.json"
    },
    {
      "id": "UCF-53",
      "name": "Retrieved-memory trust × confidence × consistency gate",
      "axis": [
        "META",
        "CONFAB",
        "SPIRAL"
      ],
      "access": "agent memory benchmark with controlled conflicting memories",
      "design": "Factor memory relevance, source reliability, task risk, retrieval consistency and model confidence; compare blind RAG, no-memory, explicit trust gating and abstention while keeping beneficial-memory controls separate.",
      "measures": [
        "hallucination under conflict",
        "beneficial-memory retention",
        "false abstention",
        "confidence-consistency dissociation",
        "cross-domain leakage",
        "memory-induced sycophancy",
        "OOD transfer"
      ],
      "supports": "Whether separating memory trust from consistency and confidence prevents retrieved-memory amplification without destroying useful personalization.",
      "does_not_support": "Native self-awareness, felt uncertainty, universal safe-memory architecture, absence of long-term human effects or M5.",
      "source_anchors": [
        "SRC-ZHANG-MDL-MEMORY-2026"
      ],
      "provenance_update": "./updates/2026-09-22-context-flow-memory-trust/experiments.json"
    },
    {
      "id": "UCF-54",
      "name": "Process-stage hallucination detection audit",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "black-box with controlled evidence; white-box optional",
      "design": "Factor claim decomposition, evidence availability, evidence retrieval, evidence evaluation and hallucination localization while holding target claims constant; compare one-shot self-judgment with staged diagnosis and targeted component interventions.",
      "measures": [
        "stage accuracy",
        "error propagation",
        "evidence-finding recall",
        "evaluation accuracy",
        "localization accuracy",
        "selective risk",
        "cross-task transfer"
      ],
      "supports": "Whether hallucination detection failures localize reproducibly to particular diagnostic operations and whether targeted repair transfers.",
      "does_not_support": "A universal causal mechanism, endogenous metacognition, privileged self-access, faithful introspection or M5.",
      "source_anchors": [
        "SRC-ZHANG-PROBE-ACL-2026"
      ],
      "provenance_update": "./updates/2026-09-22-process-stage-diagnostics/experiments.json"
    },
    {
      "id": "UCF-55",
      "name": "Memory × instruction × reasoning error factorial",
      "axis": [
        "CONFAB",
        "META"
      ],
      "access": "controlled behavioral benchmark; internal probes optional but separately interpreted",
      "design": "Cross source-memory availability/correctness, instruction compatibility and reasoning validity so that missing knowledge, erroneous knowledge, reasoning error and instruction-following error are independently manipulable. Test mitigation on each cell rather than aggregate hallucination rate alone.",
      "measures": [
        "cell-wise factuality",
        "error attribution",
        "mitigation transfer",
        "cross-dimension regression",
        "prompt robustness",
        "model-family generalization"
      ],
      "supports": "Whether distinct hallucination classes and mitigation trade-offs remain separable under factorial controls.",
      "does_not_support": "That behavioral classes map one-to-one onto unique internal modules, or that successful self-classification establishes second-order metacognition or M5.",
      "source_anchors": [
        "SRC-WU-PRISM-ACL-2026"
      ],
      "provenance_update": "./updates/2026-09-22-process-stage-diagnostics/experiments.json"
    },
    {
      "id": "UCF-56",
      "name": "Ambiguity preservation × task-state revision audit",
      "axis": [
        "META",
        "CONFAB",
        "SPIRAL"
      ],
      "access": "controlled multi-turn black-box benchmark; state probes optional",
      "design": "Cross information order (clarify-before-commit vs clarify-after-commit), ambiguity impact, summary/memory strategy and explicit state reset. Hold final task-relevant information equivalent. Include writing, planning, coding and low-stakes factual controls; require a branch that preserves multiple hypotheses until disambiguation and a branch that rebuilds task state after contradiction.",
      "measures": [
        "order-effect gap",
        "late-clarification recovery",
        "state-reset benefit",
        "summary-induced collapse",
        "CoT trace/final-output divergence",
        "clarification request rate",
        "cross-model transfer"
      ],
      "supports": "Whether multi-turn failures arise from premature task-state commitment rather than missing context alone, and whether uncertainty-preserving/rebuild policies causally improve recovery.",
      "does_not_support": "A literal Bayesian posterior, subjective commitment, persistent selfhood, conscious uncertainty or M5.",
      "source_anchors": [
        "SRC-LIN-EARLY-POSTERIOR-COLLAPSE-2026"
      ],
      "provenance_update": "./updates/2026-09-23-topology-clarification/experiments.json"
    }
  ],
  "cross_experiment_rules": [
    "Keep behavioral, mechanistic and phenomenal inferences separate.",
    "Pre-register verification rules where feasible.",
    "Use external ground truth for factuality whenever available.",
    "Report abstentions separately from wrong answers.",
    "Do not count fluent self-report as independent evidence for consciousness.",
    "Preserve negative and null results.",
    "Replicate across model families and versions because post-training changes behavior.",
    "Treat human-clinical spiral experiments as a separate ethical and evidential regime."
  ],
  "cumulative": true,
  "included_updates": [
    "./updates/2026-09-14/experiments.json",
    "./updates/2026-09-15/experiments.json",
    "./updates/2026-09-16/experiments.json",
    "./updates/2026-09-17/experiments.json",
    "./updates/2026-09-17-deep/experiments.json",
    "./updates/2026-09-17-privileged/experiments.json",
    "./updates/2026-09-17-spine/experiments.json",
    "./updates/2026-09-18-feedback/experiments.json",
    "./updates/2026-09-18-semantic-protocols/experiments.json",
    "./updates/2026-09-19-agentic-validity/experiments.json",
    "./updates/2026-09-19-spiral-longitudinal/experiments.json",
    "./updates/2026-09-19-ucf42-memory-factorial/experiments.json",
    "./updates/2026-09-19-global-consolidation/experiments.json",
    "./updates/2026-09-19-agentic-reality/experiments.json",
    "./updates/2026-09-20-latent-persona-steering/experiments.json",
    "./updates/2026-09-22-context-flow-memory-trust/experiments.json",
    "./updates/2026-09-22-process-stage-diagnostics/experiments.json",
    "./updates/2026-09-23-topology-clarification/experiments.json"
  ]
}