| { |
| "schema_version": 1, |
| "result_revision": "2026-09-14", |
| "criterion": "Official", |
| "protocol": { |
| "name": "corrected_causal_continuation", |
| "training_seeds": [0, 42, 3072], |
| "evaluation_seeds": [0, 1, 42], |
| "episodes_per_evaluation_seed": 100, |
| "history_boundary": "strictly pre-start actions only; append each policy-executed action exactly once" |
| }, |
| "statistics": "mean and sample standard deviation across training-seed means", |
| "results_percent": [ |
| { |
| "cell": "LeWM", |
| "inference": "CEM 300x30", |
| "pusht": {"mean": 74.56, "sample_std": 3.67}, |
| "cube": {"mean": 67.33, "sample_std": 1.86}, |
| "reacher": {"mean": 83.11, "sample_std": 0.96}, |
| "tworoom": {"mean": 39.67, "sample_std": 8.39}, |
| "macro": {"mean": 66.17, "sample_std": 2.67} |
| }, |
| { |
| "cell": "Inverse only", |
| "inference": "Direct", |
| "pusht": {"mean": 36.11, "sample_std": 0.19}, |
| "cube": {"mean": 67.56, "sample_std": 4.03}, |
| "reacher": {"mean": 90.56, "sample_std": 1.35}, |
| "tworoom": {"mean": 78.56, "sample_std": 6.68}, |
| "macro": {"mean": 68.19, "sample_std": 0.67} |
| }, |
| { |
| "cell": "Goal intent only", |
| "inference": "Direct", |
| "pusht": {"mean": 81.78, "sample_std": 0.96}, |
| "cube": {"mean": 100.00, "sample_std": 0.00}, |
| "reacher": {"mean": 88.67, "sample_std": 0.33}, |
| "tworoom": {"mean": 69.22, "sample_std": 7.04}, |
| "macro": {"mean": 84.92, "sample_std": 1.86} |
| }, |
| { |
| "cell": "Goal-displacement INTACT", |
| "inference": "Direct", |
| "pusht": {"mean": 86.11, "sample_std": 0.96}, |
| "cube": {"mean": 100.00, "sample_std": 0.00}, |
| "reacher": {"mean": 97.22, "sample_std": 0.84}, |
| "tworoom": {"mean": 81.56, "sample_std": 3.67}, |
| "macro": {"mean": 91.22, "sample_std": 0.51} |
| } |
| ], |
| "manifest_note": "PAPER_E5_* manifests preserve checkpoint identities and hashes; this file is authoritative for corrected-evaluator success rates." |
| } |
|
|