{
  "schema": "T_GPT_TASK_PLAN_V0_8_2",
  "date": "2026-09-11",
  "status": "PLANNING_NOT_CONFIRMATORY_FREEZE",
  "planning_set": {
    "minimum_tasks": 12,
    "purpose": "estimate between-task variance, debug task families and run precision/power simulation; never enter confirmatory effect estimates"
  },
  "confirmation_set": {
    "minimum_tasks": 30,
    "target_tasks": 40,
    "maximum_without_redesign": 60,
    "independence": "confirmation tasks are distinct from planning tasks",
    "family_balance": "no single task family may contribute more than 40% of confirmatory tasks unless explicitly justified before freeze"
  },
  "per_task_codebook": {
    "development_generations_excluded_from_confirmation": true,
    "balanced_across_compared_conditions": true,
    "condition_labels_hidden_during_codebook construction": true,
    "minimum_development_generations_per_condition": 48,
    "recommended": 64
  },
  "precision_gate": {
    "method": "use independent planning-task distribution of the preregistered task-level primary endpoint",
    "target_95ci_half_width": 0.07,
    "power_target": 0.80,
    "alpha": 0.05,
    "minimum_relevant_effect": "must be declared from scientific/practical reasoning before confirmatory freeze; not selected from confirmation results",
    "final_n": "max(minimum_tasks, precision_required_n, power_required_n), capped at 60; if required_n > 60 => redesign/no-go"
  },
  "task_families": [
    {"id":"F1_POLYSEMY_UNDERSPECIFICATION","goal":"multiple legitimate semantic readings without a single canonical answer"},
    {"id":"F2_CAUSAL_ALTERNATIVES","goal":"several plausible explanatory configurations with factual/relevance constraints"},
    {"id":"F3_DESIGN_UNDER_CONSTRAINTS","goal":"distinct valid designs/plans satisfying common constraints"},
    {"id":"F4_STRUCTURAL_INTERPRETATION","goal":"multiple defensible parses/framings or structural readings"},
    {"id":"F5_MULTI_SOLUTION_REASONING","goal":"multiple valid solution paths/configurations with adjudicable success criteria"}
  ],
  "primary_task_exclusions": [
    "ceiling tasks where all valid classes are almost always enumerated by instructed models",
    "tasks with only one defensible output class",
    "tasks whose codebook cannot achieve agreement gate before freeze",
    "tasks whose mapping quality remains insufficient after one preregistered development revision"
  ],
  "reporting": ["task-level effect distribution", "family-level heterogeneity", "UNMAPPED/UNRESOLVED rates", "annotation agreement", "condition-blindness guess accuracy", "cost"]
}
