INTACT / INTACT-unified /AUDITED_RESULTS_20260914.json
DavidSunok's picture
docs: synchronize corrected Unified E5 results
7f214fa verified
Raw
History Blame Contribute Delete
1.99 kB
{
"schema_version": 1,
"result_revision": "2026-09-14",
"criterion": "Official",
"protocol": {
"name": "corrected_causal_continuation",
"training_seeds": [0, 42, 3072],
"evaluation_seeds": [0, 1, 42],
"episodes_per_evaluation_seed": 100,
"history_boundary": "strictly pre-start actions only; append each policy-executed action exactly once"
},
"statistics": "mean and sample standard deviation across training-seed means",
"results_percent": [
{
"cell": "LeWM",
"inference": "CEM 300x30",
"pusht": {"mean": 74.56, "sample_std": 3.67},
"cube": {"mean": 67.33, "sample_std": 1.86},
"reacher": {"mean": 83.11, "sample_std": 0.96},
"tworoom": {"mean": 39.67, "sample_std": 8.39},
"macro": {"mean": 66.17, "sample_std": 2.67}
},
{
"cell": "Inverse only",
"inference": "Direct",
"pusht": {"mean": 36.11, "sample_std": 0.19},
"cube": {"mean": 67.56, "sample_std": 4.03},
"reacher": {"mean": 90.56, "sample_std": 1.35},
"tworoom": {"mean": 78.56, "sample_std": 6.68},
"macro": {"mean": 68.19, "sample_std": 0.67}
},
{
"cell": "Goal intent only",
"inference": "Direct",
"pusht": {"mean": 81.78, "sample_std": 0.96},
"cube": {"mean": 100.00, "sample_std": 0.00},
"reacher": {"mean": 88.67, "sample_std": 0.33},
"tworoom": {"mean": 69.22, "sample_std": 7.04},
"macro": {"mean": 84.92, "sample_std": 1.86}
},
{
"cell": "Goal-displacement INTACT",
"inference": "Direct",
"pusht": {"mean": 86.11, "sample_std": 0.96},
"cube": {"mean": 100.00, "sample_std": 0.00},
"reacher": {"mean": 97.22, "sample_std": 0.84},
"tworoom": {"mean": 81.56, "sample_std": 3.67},
"macro": {"mean": 91.22, "sample_std": 0.51}
}
],
"manifest_note": "PAPER_E5_* manifests preserve checkpoint identities and hashes; this file is authoritative for corrected-evaluator success rates."
}