{
  "schema": "ULTRACON_AI_EPISTEMIC_SOURCE_ADDENDUM_V1",
  "date": "2026-09-19",
  "sources": [
    {
      "id": "SRC-KARETNIKOV-HUMAN-PROXIES-2026",
      "year": 2026,
      "title": "Large language models as human proxies",
      "authors": "Nikita Karetnikov, Iyad Rahwan, Davor Svetinovic",
      "venue": "Nature Computational Science",
      "publication_date": "2026-09-18",
      "peer_reviewed": true,
      "doi": "10.1038/s43588-026-01060-3",
      "url": "https://www.nature.com/articles/s43588-026-01060-3",
      "axes": [
        "CONSC",
        "SPIRAL",
        "META"
      ],
      "evidence": [
        "PEER_REVIEWED",
        "REVIEW",
        "METHOD"
      ],
      "role": [
        "human-proxy validity",
        "construct validity",
        "behavior-mechanism separation",
        "simulation roles"
      ],
      "status": "peer_reviewed_review",
      "reported_anchor": "Review distinguishes four uses of LLMs as human proxies—believable agents, task agents, experimental subjects and silicon samples—and argues that human similarity is not a single property: each role supports different scientific claims and requires its own validity criteria.",
      "non_inference": "Human-like behavior in one role or construct does not establish human-equivalent mechanism, cognition, phenomenology, population representativeness or validity in another proxy role."
    },
    {
      "id": "SRC-IYER-TOOL-HALLUCINATION-2026",
      "year": 2026,
      "title": "Closed-World Resolution Against Tool Hallucination in LLM Agents",
      "authors": "Laxmipriya Ganesh Iyer",
      "venue": "arXiv preprint 2609.19425",
      "publication_date": "2026-09-16",
      "peer_reviewed": false,
      "arxiv": "2609.19425",
      "url": "https://arxiv.org/abs/2609.19425",
      "axes": [
        "CONFAB",
        "META"
      ],
      "evidence": [
        "PREPRINT",
        "BENCH",
        "EMP_BEHAV",
        "SYSTEMS"
      ],
      "role": [
        "tool hallucination",
        "closed-world resolution",
        "schema validity",
        "MCP namespace failure"
      ],
      "status": "exceptional_recent_preprint_not_peer_reviewed",
      "reported_anchor": "Preprint reports 322 genuine tool hallucinations across ten hosted models under two invocation surfaces; fabricated tool calls were much more frequent on unconstrained raw-JSON surfaces (34 versus 3). Extending the benchmark to merged MCP namespaces yielded 154 additional incidents, including collision/shadowing failures.",
      "non_inference": "Tool hallucination is an agent-action/interface failure and must not be collapsed into ordinary factual hallucination. The benchmark does not establish universal rates, immunity through closed-world resolution, or that model scale never helps outside the tested systems."
    }
  ]
}
