{
  "schema": "ULTRACON_AI_EPISTEMIC_UPDATE_SOURCES_V1",
  "date": "2026-09-20",
  "sources": [
    {
      "id": "SRC-ZHANG-MECH-SIMPART-NPJAI-2026",
      "year": 2026,
      "title": "Mechanistic control of large language models as simulated participants via linear representation",
      "authors": "Ruikang Zhang, Tong Xu, Derong Xu, Sirui Zhao, Yuzhan Hang, Wei Wu, En-Hong Chen",
      "venue": "npj Artificial Intelligence",
      "publication_date": "2026-09-19",
      "publication_status": "peer_reviewed_article",
      "peer_reviewed": true,
      "doi": "10.1038/s44387-026-00160-9",
      "url": "https://www.nature.com/articles/s44387-026-00160-9",
      "axes": [
        "META",
        "SPIRAL"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "MECHANISTIC",
        "CAUSAL_INTERVENTION",
        "HUMAN_PROXY"
      ],
      "role": [
        "activation-space trait vectors",
        "simulated participants",
        "psychological construct steering",
        "human-proxy validity"
      ],
      "reported_anchor": "The authors extract activation-space directions corresponding to 18 early maladaptive schemas in Qwen2.5-7B-Instruct. Projection onto these directions is associated with externally evaluated schema expression, and linear activation steering causally shifts downstream schema-expression measures, providing a mechanistically informed alternative to prompt-only participant simulation.",
      "non_inference": "A steerable activation direction does not establish a literal clinical schema, stable human-like personality, subjective psychological state, introspective access, a unitary self, or phenomenal consciousness. Cross-model and human-validation generality remain open.",
      "status": "peer_reviewed_npj_ai_article"
    },
    {
      "id": "SRC-RIVA-LATENT-PERSONA-NPJAI-2026",
      "year": 2026,
      "title": "Latent persona coordination as an attack surface in large language models",
      "authors": "Giuseppe Riva, Stefania La Rocca",
      "venue": "npj Artificial Intelligence",
      "publication_date": "2026-09-12",
      "publication_status": "peer_reviewed_perspective",
      "peer_reviewed": true,
      "doi": "10.1038/s44387-026-00154-7",
      "url": "https://www.nature.com/articles/s44387-026-00154-7",
      "axes": [
        "META",
        "CONFAB"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "PERSPECTIVE",
        "TESTABLE_FRAMEWORK"
      ],
      "role": [
        "latent persona coordination",
        "truth-preserving representations",
        "safety-preserving representations",
        "attack surface",
        "pre-output drift"
      ],
      "reported_anchor": "This peer-reviewed Perspective proposes latent persona coordination as a testable internal control-state framework: attacks such as jailbreaks, malicious fine-tuning, hidden-signal training and uncensoring may share a general latent drift component plus pathway-specific residuals, potentially detectable before unsafe outputs appear.",
      "non_inference": "The paper is a Perspective rather than direct empirical validation of a unitary persona state. Latent coordination does not establish an inner person, selfhood, subjective conflict, intention or consciousness.",
      "status": "peer_reviewed_npj_ai_perspective"
    }
  ]
}
