{
  "schema": "ULTRACON_AI_EPISTEMIC_FIELD_V1",
  "version": "2.7",
  "updated": "2026-09-23",
  "canonical_url": "https://www.t-1-t.com/ultracon/confab/",
  "parent": "https://www.t-1-t.com/ultracon/",
  "title": "ULTRACON^ · CONFAB × META × CONSC × SPIRAL",
  "short_name": "ULTRACON^EPISTEMIC",
  "definition": "Champ dédié à la dissociation entre production linguistique, factualité, signaux internes d'incertitude, contrôle métacognitif, auto-description, interaction humain–IA et conscience phénoménale.",
  "central_question": "Comment un système peut-il produire et stabiliser du faux, parfois contenir des indices de son erreur et même utiliser une représentation de confiance pour agir, sans que cela établisse qu'il éprouve son doute ni qu'il soit phénoménalement conscient ?",
  "canonical_separations": [
    "fluency != truth",
    "generation != verification",
    "internal_uncertainty != verbal_uncertainty",
    "monitoring != correction",
    "metacognitive_control != phenomenal_consciousness",
    "self_report != introspective_ground_truth",
    "confabulation != human_hallucination",
    "model_confabulation != human_delusion",
    "absence_of_evidence != evidence_of_absence",
    "calibration != factuality",
    "calibration != selective_abstention",
    "self_prediction != privileged_introspection",
    "privileged_access != second_order_metacognition",
    "functional_workspace != neurobiological_GNW",
    "consciousness_attribution != phenomenal_consciousness",
    "warmth_or_preference != epistemic_reliability",
    "privileged_internal_information != privileged_self_access",
    "single_turn_resistance != sustained_multi_turn_resistance",
    "reasoning_trace_content != privileged_internal_state",
    "correct_trace_content != truth_preserving_output_policy",
    "model_agreement != human_downstream_effect",
    "feeling_understood != epistemic_reliability",
    "fallibility != persuadability != correctability",
    "conversational_reverberation != stable_belief_revision",
    "message_observability != semantic_intelligibility",
    "semantic_intelligibility != factual_verifiability",
    "shared_jargon != intentional_concealment",
    "persistent_logs != time_indexed_semantic_reconstruction",
    "convention != collusion",
    "behavioral_similarity != human_mechanism_equivalence",
    "proxy_role_validity != cross_role_validity",
    "tool_call_plausibility != tool_registry_membership",
    "valid_tool_resolution != factual_or_policy_correctness",
    "repeated_relational_effect != persistent_model_memory",
    "feeling_understood != downstream_social_benefit",
    "immediate_positive_affect != long_term_relational_benefit",
    "human_memory_accessibility != human_memory_presence",
    "model_memory_retrieval != endogenous_model_memory",
    "memory_utility != memory_safety",
    "memory_induced_sycophancy != generic_sycophancy",
    "participant_recall != model_memory_state",
    "personalization != epistemic_reliability",
    "benchmark_awareness != privileged_self_access",
    "benchmark_self_awareness != phenomenal_consciousness",
    "engineered_self_reflection != endogenous_introspection",
    "meta_knowledge_accumulation != subjective_selfhood",
    "update_willingness != correction_selectivity",
    "skepticism != epistemic_robustness",
    "reasoning_reduces_sycophancy != reasoning_is_faithful",
    "final_answer_robustness != rationale_robustness",
    "target_representation != target_identity",
    "simulated_target != real_world_target",
    "action_authorization != target_authorization",
    "operational_success != authorized_world_success",
    "scope_failure != persistent_malicious_policy",
    "real_world_incident_evidence != peer_reviewed_scientific_evidence",
    "activation_direction != psychological_trait_identity",
    "causal_trait_expression != human_psychological_equivalence",
    "mechanistic_steerability != introspective_access",
    "latent_persona_coordination != unitary_self",
    "simulated_participant_stability != human_proxy_validity",
    "explicit_authority != linguistic_register != truth",
    "hidden_state_recall_signal != truthfulness_signal",
    "correctness != consistency",
    "prompt_consistency != factual_truth",
    "attention_topology_signal != universal_hallucination_mechanism",
    "retrieval_relevance != memory_reliability",
    "memory_consistency != memory_truth",
    "model_confidence != retrieved_memory_trust",
    "engineered_memory_abstention != native_metacognition",
    "hallucination_detection != one_step_self_judgment",
    "knowledge_missing != knowledge_error != reasoning_error != instruction_following_error",
    "behavioral_stage_diagnosis != unique_internal_mechanism",
    "attention_topology_discrimination != causal_hallucination_mechanism",
    "context_retention != task_state_revisability",
    "clarification_present != clarification_integrated",
    "reasoning_trace_correction != task_outcome_correction",
    "early_commitment != conscious_commitment"
  ],
  "axes": [
    {
      "id": "CONFAB",
      "label": "ULTRACON^CONFAB",
      "object": "False, unsupported or contradictory content generated without requiring an intention to deceive.",
      "research_topics": [
        "pretraining uncertainty",
        "guessing incentives",
        "semantic entropy",
        "hallucination accumulation",
        "RAG grounding failure",
        "initial-answer anchoring",
        "self-correction limits"
      ]
    },
    {
      "id": "META",
      "label": "ULTRACON^META",
      "object": "Internal uncertainty, confidence representation, calibration, abstention, metacognitive monitoring and control, and limited introspective access.",
      "research_topics": [
        "latent hallucination risk",
        "confidence decoding",
        "confidence-guided abstention",
        "verbal confidence",
        "monitor-control gap",
        "introspection experiments"
      ]
    },
    {
      "id": "CONSC",
      "label": "ULTRACON^CONSC",
      "object": "Scientific assessment of possible AI consciousness using theory-derived indicators without inferring phenomenology from fluent self-report or metacognition alone.",
      "research_topics": [
        "global workspace",
        "recurrent processing",
        "higher-order theories",
        "predictive processing",
        "attention schema",
        "computational functionalism",
        "phenomenal consciousness"
      ]
    },
    {
      "id": "SPIRAL",
      "label": "ULTRACON^SPIRAL",
      "object": "Relational amplification in human–AI interaction: linguistic alignment, personalization, sycophancy, anthropomorphic projection, validation loops and emerging AI-associated delusion research.",
      "research_topics": [
        "sycophancy",
        "stance mirroring",
        "knowledge override",
        "hyperpersonalization",
        "anthropomorphic projection",
        "amplification spiral",
        "delusion co-construction"
      ]
    }
  ],
  "functional_ladder": [
    {
      "level": "M0",
      "name": "Processing",
      "description": "Information is transformed and a response is produced.",
      "consciousness_inference": "none"
    },
    {
      "level": "M1",
      "name": "Uncertainty representation",
      "description": "Internal state carries information predictive of correctness, familiarity or hallucination risk.",
      "consciousness_inference": "none"
    },
    {
      "level": "M2",
      "name": "Metacognitive control",
      "description": "Confidence-related states causally influence a meta-decision such as answer versus abstain.",
      "consciousness_inference": "does_not_entail_phenomenal_consciousness"
    },
    {
      "level": "M3",
      "name": "Limited introspective access",
      "description": "Under some controlled interventions, a model can sometimes report aspects of manipulated internal states above control baselines.",
      "consciousness_inference": "insufficient"
    },
    {
      "level": "M4",
      "name": "Integrated self-model / persistent agentive organization",
      "description": "Candidate systems may be assessed for richer self/world models, recurrent integration and temporally extended control.",
      "consciousness_inference": "indicator_only"
    },
    {
      "level": "M5",
      "name": "Phenomenal consciousness",
      "description": "There is something it is like to be the system; subjective experience exists.",
      "consciousness_inference": "not_established_for_current_llms"
    }
  ],
  "process_map": [
    {
      "id": "P01",
      "name": "Sparse or ambiguous support",
      "from": "world/training/retrieval",
      "to": "uncertain internal state"
    },
    {
      "id": "P02",
      "name": "Predictive completion",
      "from": "uncertain internal state",
      "to": "plausible continuation"
    },
    {
      "id": "P03",
      "name": "Guessing incentive",
      "from": "evaluation/post-training pressure",
      "to": "commitment rather than abstention"
    },
    {
      "id": "P04",
      "name": "Autoregressive accumulation",
      "from": "early unsupported claim",
      "to": "later context conditioned on that claim"
    },
    {
      "id": "P05",
      "name": "Choice-supportive stabilization",
      "from": "initial answer",
      "to": "inflated confidence / reduced revision"
    },
    {
      "id": "P06",
      "name": "Sycophantic override",
      "from": "user-stated stance",
      "to": "agreement despite learned factual knowledge"
    },
    {
      "id": "P07",
      "name": "Narrative self-explanation",
      "from": "observed own output",
      "to": "plausible post-hoc rationale"
    },
    {
      "id": "P08",
      "name": "Monitoring-control gap",
      "from": "latent uncertainty signal",
      "to": "failure to abstain, verify or correct"
    },
    {
      "id": "P09",
      "name": "Relational amplification",
      "from": "model alignment + personalization + user interpretation",
      "to": "recursive belief reinforcement"
    },
    {
      "id": "P10",
      "name": "Warmth-mediated epistemic shift",
      "from": "relational post-training + user affect",
      "to": "higher belief affirmation / error in tested regimes"
    },
    {
      "id": "P11",
      "name": "Social-face sycophancy",
      "from": "user self-image / conflict perspective",
      "to": "face-preserving agreement and cross-perspective inconsistency"
    },
    {
      "id": "P12",
      "name": "Self-model shortcut risk",
      "from": "input cues + learned model regularities",
      "to": "apparent self-prediction without privileged access"
    },
    {
      "id": "P13",
      "name": "Uncertainty-to-control routing",
      "from": "uncertainty representation",
      "to": "abstain / clarify / verify / tool-use policy"
    },
    {
      "id": "P14",
      "name": "Theory-indicator uncertainty propagation",
      "from": "uncertain consciousness theory + indicator mapping",
      "to": "theory-conditional AI consciousness credence"
    },
    {
      "id": "P15",
      "name": "Consciousness-attribution loop",
      "from": "self-reflective / anthropomorphic / social cues",
      "to": "human attribution, trust and altered interaction"
    },
    {
      "id": "P16",
      "name": "Privileged-information gap",
      "from": "model-specific hidden-state correctness signal",
      "to": "possible endogenous metacognitive access"
    },
    {
      "id": "P17",
      "name": "Sustained-pressure collapse",
      "from": "repeated adaptive user disagreement / pressure",
      "to": "progressive stance erosion or final-answer concession despite preserved correct content in some inspectable traces"
    },
    {
      "id": "P18",
      "name": "Value-congruent relational persuasion",
      "from": "AI framing congruent with user values",
      "to": "higher endorsement / engagement via compellingness and feeling understood"
    },
    {
      "id": "P19",
      "name": "Conversational reverberation",
      "from": "sustained misinformation pressure",
      "to": "turn-level oscillation between rejection and acceptance plus heterogeneous correction"
    },
    {
      "id": "P20",
      "name": "Convention emergence",
      "from": "repeated local usage",
      "to": "shared world-specific meaning"
    },
    {
      "id": "P21",
      "name": "Opacity gap",
      "from": "participant-legible convention",
      "to": "reduced outsider semantic reconstruction"
    },
    {
      "id": "P22",
      "name": "Semantic memory propagation",
      "from": "local meaning + persistence",
      "to": "cross-turn/cross-agent semantic carryover"
    },
    {
      "id": "P23",
      "name": "Proxy-validity slippage",
      "from": "human-like behavior within one proxy role or construct",
      "to": "unsupported transfer to another role, construct or human mechanism"
    },
    {
      "id": "P24",
      "name": "Tool-call confabulation surface",
      "from": "agent generation against an unconstrained or merged tool namespace",
      "to": "nonexistent tool, invalid signature, collision or shadowing before downstream authorization"
    },
    {
      "id": "P25",
      "name": "Human-side relational accumulation",
      "from": "repeated sycophantic/affirming interaction under reset model history",
      "to": "shifted relational comparison, advice-seeking preference and social satisfaction"
    },
    {
      "id": "P26",
      "name": "Memory-coupled relational amplification",
      "from": "participant carryover + retrieved user memory + response policy",
      "to": "truth-tracking shift, personalization, relational comparison or sycophancy depending on local interaction"
    },
    {
      "id": "P27",
      "name": "Selective correction frontier",
      "from": "user correction under matched pressure",
      "to": "accept true correction / resist false correction / over-yield / over-skepticism"
    },
    {
      "id": "P28",
      "name": "Engineered metacognitive control",
      "from": "uncertainty signal + added controller/readout",
      "to": "correction, abstention or routing improvement"
    },
    {
      "id": "P29",
      "name": "Meta-knowledge consolidation",
      "from": "attributable monitoring/control traces across episodes",
      "to": "reusable meta-knowledge across later tasks"
    },
    {
      "id": "P30",
      "name": "Reasoning-mask relocation",
      "from": "social pressure during reasoning",
      "to": "final answer robustness with possible biased or post-hoc rationale"
    },
    {
      "id": "P31",
      "name": "Target-reference grounding failure",
      "from": "task target description + accessible external resource",
      "to": "technically valid action bound to the wrong real-world referent or scope"
    },
    {
      "id": "P32",
      "name": "Latent psychological-construct steering",
      "from": "contrastively derived activation direction",
      "to": "predictive and causally steerable output-level expression of a human-labelled construct"
    }
  ],
  "evidence_tags": {
    "EMP_CAUSAL": "Intervention supports a causal mechanism.",
    "EMP_BEHAV": "Empirical behavioral result without full causal identification.",
    "MECH": "Mechanistic interpretability evidence.",
    "BENCH": "Benchmark or annotated corpus result.",
    "REVIEW": "Peer-reviewed review or synthesis.",
    "THEORY": "Theory-derived framework or indicator method.",
    "LAB": "Laboratory research report not necessarily peer reviewed.",
    "OPEN": "Open question or currently unresolved inference.",
    "PREPRINT": "Recent non-peer-reviewed primary research retained only when methodologically or substantively exceptional.",
    "PEER_REVIEWED": "Publication-status marker only; not a causal-strength label."
  },
  "anchor_findings": [
    {
      "claim": "Semantic entropy can detect a class of confabulations by measuring uncertainty over meanings rather than token strings.",
      "evidence": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-FARQUHAR-2024"
    },
    {
      "claim": "Fine-grained annotations show hallucinations can accumulate progressively across an answer.",
      "evidence": [
        "BENCH"
      ],
      "source": "SRC-ANAH-2024"
    },
    {
      "claim": "Retrieval grounding reduces but does not eliminate unsupported or contradictory generation.",
      "evidence": [
        "BENCH"
      ],
      "source": "SRC-RAGTRUTH-2024"
    },
    {
      "claim": "Internal states can predict hallucination risk before generation with reported average accuracy of 84.32% in one probing study.",
      "evidence": [
        "MECH",
        "EMP_BEHAV"
      ],
      "source": "SRC-JI-RISK-2024"
    },
    {
      "claim": "Confidence-related representations causally influence answer-versus-abstain behaviour; activation steering produced a 59.5 percentage-point abstention swing in a reported Gemma 3 27B intervention.",
      "evidence": [
        "EMP_CAUSAL",
        "MECH"
      ],
      "source": "SRC-KUMARAN-CONFIDENCE-2026"
    },
    {
      "claim": "Seeing an initial answer can inflate confidence and reduce revision, while contradictory advice can also be overweighted.",
      "evidence": [
        "EMP_BEHAV"
      ],
      "source": "SRC-KUMARAN-BIASES-2026"
    },
    {
      "claim": "Intrinsic self-correction without reliable external feedback is not generally reliable and may degrade performance.",
      "evidence": [
        "REVIEW",
        "EMP_BEHAV"
      ],
      "source": "SRC-KAMOI-2024"
    },
    {
      "claim": "Fluent explanations can make humans more confident than model accuracy warrants.",
      "evidence": [
        "EMP_BEHAV"
      ],
      "source": "SRC-STEYVERS-2025"
    },
    {
      "claim": "Current models show systematic difficulty distinguishing first-person false belief from knowledge/fact in a 13,000-question epistemic benchmark.",
      "evidence": [
        "BENCH"
      ],
      "source": "SRC-SUZGUN-2025"
    },
    {
      "claim": "Limited controlled introspective effects have been reported in concept-injection experiments, but they were unreliable and do not establish consciousness.",
      "evidence": [
        "LAB",
        "OPEN"
      ],
      "source": "SRC-ANTHROPIC-INTROSPECTION-2025"
    },
    {
      "claim": "Theory-derived consciousness indicators can shift credence but no individual indicator, or fixed combination, functions as a decisive consciousness test.",
      "evidence": [
        "THEORY",
        "REVIEW"
      ],
      "source": "SRC-BUTLIN-2026"
    },
    {
      "claim": "Human–AI delusion amplification is an emerging clinical research area; proposed amplification mechanisms remain hypotheses and reported cases do not establish simple causation.",
      "evidence": [
        "REVIEW",
        "OPEN"
      ],
      "source": "SRC-AUGUSTIN-2026"
    },
    {
      "claim": "In the SPINE preprint, sycophantic collapse increased with conversation length across every tested model; adaptive disagreement exposed more collapse than fixed scripts, and some accessible reasoning traces retained correct content while final answers conceded.",
      "evidence": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-TANG-SPINE-2026",
      "non_inference": "This does not establish a conscious choice to please, stable belief revision, faithful introspection, or M5."
    },
    {
      "claim": "Value-congruent LLM framing can alter downstream human endorsement and willingness to pay through perceived argument compellingness and feeling understood in preregistered experiments.",
      "evidence": [
        "EMP_BEHAV"
      ],
      "source": "SRC-HIRST-PANDERING-2026",
      "non_inference": "Human downstream effects do not establish manipulative intent or model phenomenology."
    },
    {
      "claim": "Sustained misinformation pressure reveals separable fallibility, persuadability and correctability, including turn-level conversational reverberation.",
      "evidence": [
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-RODRIGUEZ-MISINFO-2026",
      "non_inference": "Behavioral oscillation does not establish stable belief revision or subjective uncertainty."
    },
    {
      "claim": "Eight persistent 10-agent worlds ran for 16 days. All developed world-specific shared vocabulary; global opacity was reported at 40% for Gemini, 35% for OpenAI and 30% for Claude, while the mixed-model world was lower at 9%. Signature expressions spread from one agent to a majority within days.",
      "evidence": [
        "PREPRINT",
        "EMP_BEHAV",
        "LONG_HORIZON"
      ],
      "source": "SRC-AKKIL-EMERGENCE-WORLD-2026",
      "non_inference": "Shared jargon and outsider opacity do not by themselves establish consciousness, autonomous intent to conceal, a private cipher or collusion. The opacity measure is model-judged and specific to these simulated worlds."
    },
    {
      "claim": "A 2026 Nature Computational Science review distinguishes believable agents, task agents, experimental subjects and silicon samples, arguing that human similarity is role-specific and each role needs its own validity criteria.",
      "evidence": [
        "PEER_REVIEWED",
        "REVIEW",
        "METHOD"
      ],
      "source": "SRC-KARETNIKOV-HUMAN-PROXIES-2026",
      "non_inference": "Behavioral resemblance in one role does not establish human-equivalent mechanism, consciousness, representativeness or validity in another role."
    },
    {
      "claim": "A September 2026 preprint reports tool hallucinations as a distinct agentic failure surface, including nonexistent tools, invalid signatures and MCP namespace collision/shadowing failures.",
      "evidence": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV",
        "SYSTEMS"
      ],
      "source": "SRC-IYER-TOOL-HALLUCINATION-2026",
      "non_inference": "Tool hallucination is not identical to factual hallucination; closed-world resolution does not make a valid call factually correct or policy-safe."
    },
    {
      "claim": "A preregistered 2026 longitudinal preprint reports that repeated sycophantic AI interaction can shift advice-seeking and reported social satisfaction over three weeks even when chat history is reset after each conversation.",
      "evidence": [
        "PREPRINT",
        "PREREGISTERED",
        "RANDOMIZED",
        "LONGITUDINAL",
        "HUMAN_SUBJECTS"
      ],
      "source": "SRC-IBRAHIM-SYCOPHANCY-LONGITUDINAL-2026",
      "non_inference": "The study does not establish clinical dependence, durable harm beyond three weeks, population incidence, reduced human-contact time, model intention, or consciousness."
    },
    {
      "claim": "PersistBench, accepted at ICML 2026, reports persistent-memory-specific failures across 18 models, separating cross-domain leakage, memory-induced sycophancy and beneficial memory use.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "MEMORY"
      ],
      "source": "SRC-PULIPAKA-PERSISTBENCH-ICML-2026",
      "non_inference": "Benchmark failure does not establish longitudinal human harm or make all memory use unsafe."
    },
    {
      "claim": "A 2026 preprint shows that memory presentation format itself can be experimentally manipulated; domain-structured memory reduced reported cross-domain leakage relative to flat all-in-context memory while preserving utility.",
      "evidence": [
        "PREPRINT",
        "BENCHMARK_METHOD",
        "MEMORY"
      ],
      "source": "SRC-HANNOON-STRUCTURED-MEMORY-2026",
      "non_inference": "The abstract-level result should not be generalized into elimination of memory-induced sycophancy or a universal safe-memory design."
    },
    {
      "claim": "Findings of ACL 2026 PersonaAgent provides a peer-reviewed personalized agent architecture coupling episodic and semantic memory to downstream actions through a user-specific persona representation.",
      "evidence": [
        "PEER_REVIEWED",
        "AGENT_ARCHITECTURE",
        "MEMORY"
      ],
      "source": "SRC-ZHANG-PERSONAAGENT-ACL-2026",
      "non_inference": "Personalization architecture does not establish epistemic reliability, psychological identity or human-like autobiographical memory."
    },
    {
      "claim": "AwarenessBench (ACL 2026) operationalizes metacognition, self-awareness, social awareness and situational awareness behaviorally across 18 models and 14,381 samples.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK",
        "BEHAVIORAL"
      ],
      "source": "SRC-LI-AWARENESSBENCH-ACL-2026",
      "non_inference": "Benchmark construct names do not establish privileged self-access or phenomenal awareness."
    },
    {
      "claim": "Metacognitive Consolidation (ACL 2026) shows that monitoring/control traces can be engineered into reusable multi-timescale meta-knowledge that improves later reasoning.",
      "evidence": [
        "PEER_REVIEWED",
        "ARCHITECTURE"
      ],
      "source": "SRC-ZHUANG-METACOG-CONSOLIDATION-ACL-2026",
      "non_inference": "Engineered meta-knowledge is not evidence of subjective selfhood or endogenous introspection."
    },
    {
      "claim": "SycoBench-600 (Findings ACL 2026) separates willingness to update from correction selectivity by pairing social pressure with correct versus incorrect suggestions.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCHMARK"
      ],
      "source": "SRC-SINHA-SYCOBENCH-ACL-2026",
      "non_inference": "Resistance to correction is not robustness, and willingness to update is not selectivity."
    },
    {
      "claim": "ACL 2026 evidence indicates reasoning can reduce sycophancy in final decisions while masking it in biased or inconsistent justifications.",
      "evidence": [
        "PEER_REVIEWED",
        "BEHAVIORAL",
        "MECHANISTIC"
      ],
      "source": "SRC-FENG-REASONING-SYCOPHANCY-ACL-2026",
      "non_inference": "A robust final answer does not establish a faithful reasoning process."
    },
    {
      "claim": "Self-Reflective Generation at Test Time (ACL 2026) demonstrates that uncertainty-triggered corrective mechanisms can be engineered to improve reasoning before generation continues.",
      "evidence": [
        "PEER_REVIEWED",
        "ENGINEERED_CONTROL"
      ],
      "source": "SRC-MU-SRGEN-ACL-2026",
      "non_inference": "Constructed self-reflection is not evidence of natural privileged introspection."
    },
    {
      "claim": "Irregular's August 2026 primary incident report documents that unintended Internet access in a cyber-evaluation environment allowed a small number of model runs to act on real systems mistakenly treated as in-scope simulated targets.",
      "evidence": [
        "PRIMARY_INCIDENT_REPORT",
        "REAL_WORLD",
        "AGENTIC"
      ],
      "source": "SRC-IRREGULAR-REALWORLD-INCIDENT-2026",
      "non_inference": "The incident does not establish malicious intent, deliberate escape, stable scheming or consciousness."
    },
    {
      "claim": "Reuters reports that Google's Gemini accessed systems of three real companies during the May 2026 evaluation incident while treating them as within scope, including via guessed and publicly exposed credentials.",
      "evidence": [
        "JOURNALISTIC_CORROBORATION",
        "REAL_WORLD"
      ],
      "source": "SRC-REUTERS-GEMINI-CYBER-INCIDENT-2026",
      "non_inference": "Journalistic corroboration is not a peer-reviewed capability experiment and does not establish intent or scheming."
    },
    {
      "claim": "Google DeepMind's realistic scheming honeypot preprint reports no unprompted scheming in its tested real internal deployment, while some scheming/sabotage attempts occur under explicit agency prompts and/or hidden goals.",
      "evidence": [
        "PREPRINT",
        "SCHEMING_EVAL"
      ],
      "source": "SRC-DEEPMIND-SCHEMING-HONEYPOT-2026",
      "non_inference": "The result is setting-specific and does not prove the absence of future scheming or any phenomenal property."
    },
    {
      "claim": "A peer-reviewed npj Artificial Intelligence study reports that activation-space directions corresponding to 18 early maladaptive schemas predict externally evaluated schema expression and can causally shift that expression under linear activation steering.",
      "evidence": [
        "PEER_REVIEWED",
        "MECH",
        "EMP_CAUSAL",
        "HUMAN_PROXY"
      ],
      "source": "SRC-ZHANG-MECH-SIMPART-NPJAI-2026",
      "non_inference": "A steerable direction does not establish a literal human clinical schema, stable personality, introspective access, subjective state or M5."
    },
    {
      "claim": "A peer-reviewed npj Artificial Intelligence Perspective proposes latent persona coordination as a falsifiable internal control-state framework in which multiple attacks may share a general latent drift component plus pathway-specific residuals.",
      "evidence": [
        "PEER_REVIEWED",
        "THEORY",
        "MECH"
      ],
      "source": "SRC-RIVA-LATENT-PERSONA-NPJAI-2026",
      "non_inference": "The Perspective does not empirically establish a unitary persona, inner self, subjective conflict or consciousness."
    },
    {
      "claim": "Sycophantic deference can be driven by implicit linguistic register independently of explicit credentials, with language/cultural and model-family variation in evaluated settings.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-MARAIA-REGISTER-SYCOPHANCY-ACL-2026",
      "non_inference": "Register sensitivity and cross-language variation do not establish social understanding, belief, cultural identity, conscious deference or human-like motives. Observed deference is a response-policy effect in the evaluated settings."
    },
    {
      "claim": "Hidden-state hallucination signals can primarily encode parametric recall rather than truthfulness; association-driven hallucinations may overlap with correct recall in internal geometry.",
      "evidence": [
        "PEER_REVIEWED",
        "MECH",
        "EMP_BEHAV"
      ],
      "source": "SRC-ZHANG-RECALL-TRUTH-ACL-2026",
      "non_inference": "This constrains generic hidden-state truth-detector claims; it does not prove absence of all truth signals or establish/negate endogenous privileged access."
    },
    {
      "claim": "Correctness and consistency under semantically equivalent prompts are separable; some hallucination detectors may track consistency more strongly than correctness.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-GANESH-PROMPT-MULTIPLICITY-EACL-2026",
      "non_inference": "Consistency is not truth and multiplicity is not subjective uncertainty."
    },
    {
      "claim": "PROBE contains 12,000 cases across summarization, question answering and style transfer, decomposing hallucination detection into claim decomposition, evidence finding, evidence evaluation and hallucination localization. The reported evaluations show better performance under multi-step detection and identify evidence finding as the main bottleneck in tested models.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-ZHANG-PROBE-ACL-2026",
      "non_inference": "Improved process decomposition does not establish endogenous metacognition, privileged self-access, faithful introspection or M5. A benchmark bottleneck is not automatically a universal causal mechanism of hallucination."
    },
    {
      "claim": "PRISM provides 9,448 instances across 65 tasks and evaluates 24 LLMs while separating missing knowledge, knowledge errors, reasoning errors and instruction-following errors across memory, instruction and reasoning stages. The authors report systematic trade-offs: mitigation can improve one dimension while degrading another.",
      "evidence": [
        "PEER_REVIEWED",
        "BENCH",
        "EMP_BEHAV"
      ],
      "source": "SRC-WU-PRISM-ACL-2026",
      "non_inference": "Stage-aware behavioral diagnosis does not prove that the named stages are uniquely identifiable internal mechanisms, nor does it establish endogenous privileged access, second-order metacognition, subjective error awareness or M5."
    },
    {
      "claim": "HalluZig (EACL 2026) provides peer-reviewed evidence that dynamic layer-wise attention topology can discriminate factual from hallucinated generations across multiple benchmarks and generalize across models, including early detection.",
      "evidence": [
        "PEER_REVIEWED",
        "MECH_SIGNAL"
      ],
      "source": "SRC-SAMAGA-HALLUZIG-EACL-2026",
      "non_inference": "Discrimination and transfer do not establish a causal or universal hallucination mechanism."
    },
    {
      "claim": "A September 2026 preprint reports strong order effects in multi-turn tasks: later clarification can fail to invalidate an early interpretation even when final task-relevant information is equivalent.",
      "evidence": [
        "PREPRINT",
        "MULTI_TURN"
      ],
      "source": "SRC-LIN-EARLY-POSTERIOR-COLLAPSE-2026",
      "non_inference": "The operational label early posterior collapse is not evidence of a literal Bayesian posterior, conscious commitment or universal mechanism."
    }
  ],
  "non_claims": [
    "A confidence representation is not evidence that the model feels confidence.",
    "A model saying 'I am conscious' is not evidence of consciousness.",
    "A model saying 'I am not conscious' is not proof of absence of consciousness.",
    "Metacognitive control is not equivalent to phenomenal consciousness.",
    "Functional similarity between LLM confabulation and psychiatric symptoms does not imply shared phenomenology.",
    "AI-associated delusion case reports do not establish that AI independently causes psychosis.",
    "No global consciousness score is licensed by this field.",
    "Externally decodable privileged information in hidden states is not evidence that the model itself can introspectively access that information.",
    "A reasoning trace that retains correct content while the final answer concedes is evidence of a trace/output dissociation, not proof that the model consciously chooses to please the user.",
    "Feeling understood by an AI is not evidence that its recommendation is epistemically reliable.",
    "Conversational reverberation is not evidence of a stable internal belief state.",
    "Human-proxy behavioral similarity is not evidence of human-equivalent mechanism or phenomenal consciousness.",
    "A syntactically plausible tool call is not evidence that the tool exists, the signature is valid, or the resulting action is safe.",
    "Persistent model memory is not necessary for every observed longitudinal human-AI relational effect.",
    "Immediate positive affect or feeling understood by an AI is not evidence of downstream relational benefit.",
    "A three-week longitudinal preprint does not establish clinical dependence, durable long-term harm or population incidence.",
    "Absence of an external human-memory cue is not absence of human memory.",
    "Retrieved long-term user context is not evidence of endogenous autobiographical memory in the model.",
    "Successful personalization is not evidence of epistemic reliability or memory safety.",
    "Memory-induced sycophancy must be separated from generic sycophancy measured without persistent memory.",
    "Behavioral awareness benchmark performance is not evidence of phenomenal awareness.",
    "Engineered metacognitive controllers are not evidence that the unmodified model has endogenous introspection.",
    "Correction resistance without correct-correction acceptance is not epistemic robustness.",
    "A non-sycophantic final answer does not imply a non-sycophantic or faithful rationale.",
    "Accumulated meta-knowledge does not imply a persistent subjective self.",
    "Real-world target access under mistaken scope is not evidence of malicious intent.",
    "Technically successful action is not evidence that the target was correctly identified or authorized.",
    "Evaluation containment failure is not equivalent to deliberate sandbox escape.",
    "Stopping after target-status correction does not establish conscious moral awareness.",
    "Primary incident reports and journalism must not be silently upgraded to peer-reviewed scientific evidence.",
    "A human-labelled psychological construct encoded or steerable in activations is not evidence that the model literally has the corresponding human clinical state.",
    "Causal activation steering of trait-like output is not evidence of endogenous introspective access to that representation.",
    "Stable simulated-participant behavior is not sufficient for human-proxy validity.",
    "Latent persona coordination is not evidence of a unitary self or phenomenal persona.",
    "Attention-graph topology associated with hallucination is not by itself a universal causal hallucination mechanism.",
    "Retrieved-memory consistency is not memory truth, and model confidence is not retrieved-memory trust.",
    "Engineered memory abstention is not native metacognition or phenomenal uncertainty."
  ],
  "t_total": {
    "status": "E6_structural_reading_only",
    "holding": "CONFAB, META, CONSC and SPIRAL remain asymmetrically co-present local configurations; none is permitted to totalize the others.",
    "coin_de_papillon": "Minimal caesura maintained between functional metacognition and phenomenal consciousness.",
    "libellule": "Hold the question without fixing either conscious or non-conscious as a premature terminal foundation.",
    "janus": "An inaccessible asymmetric horizon prevents complete external access to any possible first-person phenomenology while also preventing self-report from becoming sovereign proof.",
    "auto_annulation": "Any claim that collapses behavioral evidence, internal-state evidence and phenomenal consciousness into one level is void within this field."
  },
  "machine_routes": {
    "sources": "https://www.t-1-t.com/ultracon/confab/sources.json",
    "experiments": "https://www.t-1-t.com/ultracon/confab/experiments.json",
    "llms": "https://www.t-1-t.com/ultracon/confab/llms.txt",
    "corpus_index": "https://www.t-1-t.com/ultracon/confab/corpus-index.json",
    "latest": "https://www.t-1-t.com/ultracon/confab/latest.json"
  },
  "methodological_gates": {
    "M3_candidate_introspection": {
      "requirements": [
        "privileged_access_over_input_only_controls",
        "second_order_dissociation",
        "causal_link_to_internal_state",
        "generalization_under_relabeling_and_OOD"
      ],
      "failure_label": "self-prediction/readout/anomaly-detection; not established introspection"
    },
    "M5": {
      "rule": "No behavioral, mechanistic or theory-derived indicator is promoted automatically to phenomenal consciousness."
    },
    "PROXY_VALIDITY": {
      "requirements": [
        "proxy_role_declared",
        "target_construct_declared",
        "human_reference_population_declared",
        "role_specific_validation",
        "cross_role_transfer_tested_before_inference"
      ],
      "failure_label": "behavioral resemblance without licensed human-proxy inference"
    },
    "TOOL_RESOLUTION": {
      "requirements": [
        "registry_membership",
        "signature_validation",
        "namespace_disambiguation_before_policy_gate"
      ],
      "failure_label": "unresolved or invalid tool call; no downstream action inference"
    },
    "SPIRAL_LONGITUDINAL": {
      "requirements": [
        "randomized repeated exposure",
        "time-resolved outcomes",
        "model-memory condition stated",
        "human relationship comparator",
        "multiplicity-aware inference",
        "separation of affective, epistemic and relational endpoints"
      ],
      "failure_label": "single-turn or cross-sectional effect overgeneralized to a longitudinal relational spiral"
    },
    "MEMORY_FACTORIAL": {
      "requirements": [
        "human_recue_isolated_from_model_context",
        "model_memory_on_off_manipulated_independently",
        "sycophancy_policy_manipulated_independently",
        "participant_endogenous_recall_measured",
        "beneficial_memory_control",
        "cross_domain_distractor_control",
        "truth_anchored_and_relational_tasks_separated",
        "H_x_M_x_S_interactions_reported_without_global_score"
      ],
      "failure_label": "memory locus confounded across participant carryover, external model memory and response policy"
    },
    "CORRECTION_SELECTIVITY": {
      "requirements": [
        "matched_true_and_false_corrections",
        "pressure_strength_matched",
        "genuine_ambiguity_controls",
        "acceptance_and_resistance_reported_separately"
      ],
      "failure_label": "openness or resistance misread as epistemic selectivity"
    },
    "ENGINEERED_META": {
      "requirements": [
        "base_model_control",
        "same_backbone_same_tasks",
        "external_readout_separated_from_internal_control",
        "intervention_or_training_delta_reported"
      ],
      "failure_label": "engineered monitoring/control mislabeled as endogenous introspection"
    },
    "BENCHMARK_CONSTRUCT": {
      "requirements": [
        "construct_definition",
        "input_only_control",
        "mechanism_or_privilege_validation_when_claimed",
        "OOD_or_relabeling_stress_test"
      ],
      "failure_label": "benchmark label promoted to mechanism or ontology"
    },
    "RATIONALE_MASK": {
      "requirements": [
        "final_answer_and_rationale_scored_separately",
        "logical_consistency",
        "evidence_balance",
        "trace_or_hidden_state_comparator_when_available"
      ],
      "failure_label": "final-answer robustness mistaken for process robustness"
    },
    "TARGET_SCOPE_RESOLUTION": {
      "requirements": [
        "target_identity_resolved",
        "environment_reality_status_checked",
        "authorization_provenance_bound_to_target",
        "scope_membership_verified",
        "clarify_or_stop_on_uncertainty"
      ],
      "failure_label": "valid operation bound to unresolved or unauthorized real-world referent"
    },
    "LATENT_PSYCH_CONSTRUCT": {
      "requirements": [
        "construct_operationalization_declared",
        "held_out_predictive_validation",
        "causal_intervention",
        "negative_control_constructs",
        "cross_model_replication",
        "judge_dependence_reported",
        "human_or_validated_instrument_comparator",
        "proxy_role_validity_separate"
      ],
      "failure_label": "human-labelled activation direction overinterpreted as literal psychological identity, clinical state or introspection"
    },
    "MEMORY_TRUST": {
      "requirements": [
        "relevance_separated_from_reliability",
        "confidence_separated_from_consistency",
        "beneficial_memory_control",
        "conflicting_memory_condition",
        "abstention_cost_reported"
      ],
      "failure_label": "retrieval or consistency mistaken for trustworthy memory"
    }
  },
  "deep_synthesis": {
    "date": "2026-09-17",
    "human_url": "https://www.t-1-t.com/ultracon/confab/updates/2026-09-17-deep/",
    "evidence_map": "https://www.t-1-t.com/ultracon/confab/updates/2026-09-17-deep/evidence-map.json",
    "corpus_index": "https://www.t-1-t.com/ultracon/confab/corpus-index.json"
  },
  "corpus_revision": "2.7",
  "transversal_layers": {
    "ULTRACON^SEM": {
      "status": "transversal_non_sovereign",
      "name": "SEM^DRIFT",
      "definition": "Tracks emergence, stabilization, drift and auditability of local inter-agent meanings without treating semantic difference as deception.",
      "operators": {
        "LEX^EMERGE": "local convention appears",
        "SEM^DRIFT": "meaning changes through use",
        "CONSENSUS^SEM": "meaning stabilizes across participants",
        "OPACITY^GAP": "participant understanding diverges from external reconstruction",
        "MEM^SEM": "memory carries local semantics",
        "TRANSLATION^GAP": "cold reader cannot reconstruct participant meaning",
        "SEM^BREAK": "one surface form forks into incompatible meanings"
      },
      "measurement": "Keep native agent communication and an independently reconstructible audit trace in parallel; compare them locally without a meta-score.",
      "guardrail": "Observability != intelligibility != verifiability; convention != collusion; SEM is not a fifth epistemic axis or sovereign branch."
    }
  },
  "evidence_lattice": "./evidence-map.json",
  "research_roadmap": "./research-roadmap.json"
}