{
  "scope": "Resolve SHC teacher control failures with longer training episodes. All prior results are retained. Testing remains the standard 500-step episode.",
  "diagnosis": "In finish_v1, all eight truthful policies completed all near tests, but two failed 4/100 and 2/100 lean tests. Extending the same development evaluations to 1,000 steps exposes rail drift in seven of eight policies. Train for 2,000 steps to make delayed drift visible; keep the learned rewards and learner unchanged.",
  "change": {
    "sigma_schedule": [
      0.005,
      0.015,
      0.05,
      0.15,
      0.5,
      1.5
    ],
    "budget": 15360000,
    "starts_per_comparison": 8,
    "training_start_mix": "Four near-upright states uniform[-.05,.05]^4 and four alternating-sign leans of 6-10 degrees with cart position uniform[-.1,.1], zero linear/angular velocity. Shared incumbent/candidate starts. Each comparison draws fresh states; no test states used.",
    "training_cap": 2000
  },
  "unchanged": "Same physics and failure boundaries, 500-step test cap, mixed eight training starts, normalized five-feature sign policy, six proposal scales, strict summed return improvement, no restarts or copying. Teacher data and reward fitting are exactly v3. Larger training cap and proportionate transition budget are the only changes from finish_v1.",
  "development_seeds": [
    72000,
    72001,
    72002,
    72003,
    72004,
    72005,
    72006,
    72007
  ],
  "development_test_seed": 972000,
  "confirmation_seeds": [
    107300,
    107301,
    107302,
    107303,
    107304,
    107305,
    107306,
    107307,
    107308,
    107309,
    107310,
    107311,
    107312,
    107313,
    107314,
    107315
  ],
  "confirmation_test_seed": 9107300,
  "arms": [
    "teacher_truthful",
    "teacher_reversed",
    "teacher_none"
  ],
  "confirm_after": "All eight development teacher policies complete and stay upright in all 100 near and all 100 lean held-out tests. If this does not pass, preserve outcomes and diagnose; do not launch fresh confirmation automatically.",
  "confirmation_requirement": "All sixteen new teacher policies complete and stay upright in all 100 near and all 100 lean tests each. Record exact counts, not an unbounded guarantee. Compare all paired controls on uprightness and completion; no requirement that a truthful controller strictly beat a control already at 100% completion.",
  "old_gate": "The previous seven-of-eight strict completion superiority requirement was not met and remains recorded. This new study asks an absolute control-reliability question, separately from whether learned preferences cause upright behavior.",
  "resource_policy": "One sequential worker; BLAS one thread; retain only current run data in memory. No changes to other active queues.",
  "provenance": {
    "opening_snapshot.py": "93ba6ad5dc0a4a07d374ed2bd14d1a10e6e82fac4301cebd0948fb1f1ca7d266",
    "opening_snapshot_protocol.json": "d68c09bead9912d19ce418345204100932572ed93ca7a9356d00ec250f96af0c",
    "diagnosis.json": "9d416dd4eae3fb43e7d5af8d3126140a5940137ce85105b4954c984f01d25857",
    "previous_run": "social_cartpole_shc_teacher_finish_v1",
    "longer_episode_diagnosis_sha256": "79396fc46762a776316c22068c46d1cc98a75cb0813db6dab61763c289bddd21"
  }
}
