{ "schema_version": 1, "result_revision": "2026-09-14", "criterion": "Official", "protocol": { "name": "corrected_causal_continuation", "training_seeds": [0, 42, 3072], "evaluation_seeds": [0, 1, 42], "episodes_per_evaluation_seed": 100, "history_boundary": "strictly pre-start actions only; append each policy-executed action exactly once" }, "statistics": "mean and sample standard deviation across training-seed means", "results_percent": [ { "cell": "LeWM", "inference": "CEM 300x30", "pusht": {"mean": 74.56, "sample_std": 3.67}, "cube": {"mean": 67.33, "sample_std": 1.86}, "reacher": {"mean": 83.11, "sample_std": 0.96}, "tworoom": {"mean": 39.67, "sample_std": 8.39}, "macro": {"mean": 66.17, "sample_std": 2.67} }, { "cell": "Inverse only", "inference": "Direct", "pusht": {"mean": 36.11, "sample_std": 0.19}, "cube": {"mean": 67.56, "sample_std": 4.03}, "reacher": {"mean": 90.56, "sample_std": 1.35}, "tworoom": {"mean": 78.56, "sample_std": 6.68}, "macro": {"mean": 68.19, "sample_std": 0.67} }, { "cell": "Goal intent only", "inference": "Direct", "pusht": {"mean": 81.78, "sample_std": 0.96}, "cube": {"mean": 100.00, "sample_std": 0.00}, "reacher": {"mean": 88.67, "sample_std": 0.33}, "tworoom": {"mean": 69.22, "sample_std": 7.04}, "macro": {"mean": 84.92, "sample_std": 1.86} }, { "cell": "Goal-displacement INTACT", "inference": "Direct", "pusht": {"mean": 86.11, "sample_std": 0.96}, "cube": {"mean": 100.00, "sample_std": 0.00}, "reacher": {"mean": 97.22, "sample_std": 0.84}, "tworoom": {"mean": 81.56, "sample_std": 3.67}, "macro": {"mean": 91.22, "sample_std": 0.51} } ], "manifest_note": "PAPER_E5_* manifests preserve checkpoint identities and hashes; this file is authoritative for corrected-evaluator success rates." }