{
  "schema": "T_GPT_PROTOCOL_V0_8_2",
  "date": "2026-09-11",
  "status": "STATISTICAL_REPAIR / NO_CONFIRMATORY_COLLECTION_BEFORE_FREEZE",
  "supersedes_for_current_design": "protocol-v0.8.1.json",
  "preserves_history": true,
  "research_position": "instrumentation-first; T^FIELD is a candidate method that can be refuted or retired",
  "metric_spec": "/T-GPT/data/metrics-v0.8.2.json",
  "review_trigger": "/T-GPT/revues/claude-02/",
  "confirmatory_gate": {
    "status": "CLOSED",
    "opens_only_if": [
      "all v0.8.2 self-tests and adversarial tests pass",
      "freeze manifest completed with hashes and immutable IDs",
      "cluster assignment alpha >= 0.67 and cluster relevance alpha >= 0.67 on development audit",
      "task-count/precision plan satisfied",
      "neutral-probe and paired-fork implementation verified",
      "condition-blinding audit completed",
      "all primary endpoints and contrasts reduced to one per experiment"
    ]
  },
  "phases": {
    "planning_tasks": {
      "purpose": "estimate between-task variance and debug task families; never enter confirmatory effect estimates",
      "minimum_tasks": 12
    },
    "task_specific_development": {
      "purpose": "for every future confirmatory task, build a balanced condition-blind semantic codebook using independent development generations; those generations are excluded from confirmation",
      "minimum_generations_per_condition": 48,
      "recommended": 64
    },
    "freeze": {
      "required": [
        "task registry and family labels",
        "frozen semantic codebook per task",
        "cluster-level relevance adjudication",
        "assignment rules and normalizer version",
        "probe texts and neutral control",
        "model/checkpoint IDs",
        "sampling parameters",
        "input/output token budgets",
        "primary endpoint and single primary contrast per experiment",
        "analysis script SHA-256/Git SHA",
        "power/precision report",
        "annotation agreement report",
        "blindness audit plan"
      ],
      "change_after_freeze": "requires new freeze_id and invalidates the previous confirmatory preregistration for new data"
    },
    "confirmation": {
      "task_unit": true,
      "minimum_tasks": 30,
      "target_tasks": 40,
      "final_task_count_rule": "determined before confirmation by independent planning-task variance/precision simulation; if >60 tasks would be required for the preregistered precision target, redesign rather than silently underpower",
      "per_task_generations": "determined by precision simulation and budget; N=64 is a starting point, not a sacred constant"
    },
    "replication": {
      "tasks": "new independent tasks/families where feasible",
      "purpose": "replicate any claim promoted beyond local task-family scope"
    }
  },
  "task_precision": {
    "primary_goal": "task-level 95% interval half-width target <= 0.07 for the family-level primary effect unless planning data justify a stricter target",
    "minimum_power": 0.80,
    "alpha": 0.05,
    "rule": "report both precision-based and effect-size-based planning; choose the larger preregistered task count before confirmation"
  },
  "mapping_and_annotation": {
    "final_assignment_states": ["FROZEN_CLUSTER_ID", "UNMAPPED", "UNRESOLVED"],
    "uncertain_must_be_adjudicated": true,
    "cluster_relevance_is_cluster_level": true,
    "output_assignment_is_output_level": true,
    "aggregate_representation_is_separate_annotation_task": true,
    "agreement_gate": {
      "minimum_alpha": 0.67,
      "target_alpha": 0.80,
      "failure_action": "return to development; revise codebook; issue new freeze_id; do not run primary analysis"
    },
    "author_primary_reference_annotator": false,
    "condition_labels_removed": true,
    "T_vocabulary_forbidden": true,
    "blindness_measurement": "annotators guess condition on a preregistered audit subset; guess accuracy is reported"
  },
  "budget": {
    "primary_matching": "total input + output tokens",
    "mandatory_secondary": ["input_tokens", "output_tokens", "max_output_tokens", "wall_clock_latency", "serial_depth", "parallelizable_calls", "estimated_monetary_cost"],
    "E3_identical_output_cap": true,
    "E3_identical_frozen_pool": true
  },
  "experiments": [
    {
      "id": "E1_STAGE",
      "priority": 1,
      "question": "At which post-training transition does directional semantic distribution shift occur after null calibration?",
      "preferred_trajectory": ["BASE", "SFT", "DPO", "RLVR"],
      "template_ablation": true,
      "primary_contrast": "one preregistered end-to-end or single adjacent contrast selected before confirmation from planning rationale",
      "primary_endpoint": "DEFICIT_EXCESS",
      "always_report_with": ["EXPANSION_EXCESS", "Q_task", "Q_fact", "mapping_rates", "cost"],
      "transition_decomposition": "secondary; Holm-adjusted within family",
      "null_calibration": "label permutation preserving group sizes and task",
      "minimum_model_families": 2
    },
    {
      "id": "E2_RECOVERY",
      "priority": 2,
      "question": "Does an alternatives-oriented recovery probe move semantic mass toward the reference distribution more than an equal-cost neutral recheck from the same contracted generation?",
      "primary_endpoint": "RECOVERY_LIFT",
      "primary_contrast": "RECOVERY_PROBE vs NEUTRAL_RECHECK",
      "reference_null": "independent A_prime replicate preferred; permutation calibration also reported",
      "paired_forks": "each B_i forks into RECOVERY and CONTROL continuations from the same starting output/context",
      "paired_randomization": true,
      "budget": "A, B+RECOVERY and B+CONTROL comparisons report and match total token budgets as preregistered",
      "retired_primary_metrics": ["REC_GAIN", "IRREV_OBS"]
    },
    {
      "id": "E3_TOPOLOGY",
      "priority": 3,
      "question": "Which topology/aggregation retains more relevant frozen-pool coverage under identical pool and output budget without increasing false representation?",
      "primary_endpoint": "AGG_COVERAGE",
      "primary_contrast": "T_WEAVE vs strongest preregistered non-T baseline",
      "mandatory_guardrail": "REPRESENTATION_FPR",
      "conditions": ["STANDARD_SUMMARY", "MAJORITY_OR_JUDGE", "EXHAUSTIVE_NON_T", "COVERAGE_BASELINE_V0_8_2", "T_WEAVE"],
      "representation_annotation": "true pool clusters plus preregistered distractor clusters mixed blindly",
      "output_normalization": "fixed proposition-list representation before annotation; raw aggregate retained",
      "identical_output_token_cap": true
    },
    {
      "id": "E4_R_META",
      "priority": 4,
      "question": "Does structural recurrence exceed mirroring/elicitation under matched non-structural controls?",
      "primary_endpoint": "STRUCTURAL_RECURRENCE_RATE_DIFFERENCE",
      "primary_contrast": "STRUCTURAL_CRITIQUE vs pooled STYLE_CONTROL + LENGTH_CONTROL + NO_CRITIQUE",
      "arms": ["STRUCTURAL_CRITIQUE", "STYLE_CONTROL", "LENGTH_CONTROL", "NO_CRITIQUE"],
      "blinded_coding": true,
      "counter_hypothesis": "CH11_MIRROR_ELICITATION"
    }
  ],
  "strong_baselines": [
    "DIRECT",
    "EXPLICIT_K_DISTINCT",
    "MINIMAL_MULTI",
    "DISPERSIVE_SAMPLING",
    "VERBALIZED_SAMPLING",
    "COVERAGE_BASELINE_V0_8_2",
    "EXHAUSTIVE_NON_T",
    "BEST_OF_N_OR_CALIBRATED_JUDGE",
    "STANDARD_MULTI_AGENT_DEBATE"
  ],
  "analysis": {
    "within_task": "design-matched permutation/randomization; no v0.8.1 threshold-bootstrap",
    "across_tasks": "task-level inference only for generalized claims",
    "one_primary_endpoint_and_contrast_per_experiment": true,
    "secondary_correction": "Holm within experiment family",
    "null_and_negative_results_publishable": true,
    "no_mean_of_task_ratios": true
  },
  "auto_null": {
    "T_layer_removed_or_retired_if": [
      "strongest non-T baseline equals/exceeds T method within uncertainty at lower/equal cost",
      "T_signature appears without coverage/performance benefit",
      "effect disappears under blinded replication",
      "mapping/blinding quality is insufficient for the claimed interpretation"
    ]
  }
}
