{
 "$schema": "https://botlandscape.com/result-intervals.schema.json",
 "schema_version": 1,
 "method": {
  "version": "result-intervals/2",
  "prior": "Uniform Beta(1, 1) on the success probability.",
  "interval": "95% equal-tailed Beta(1 + k, 1 + n - k) posterior interval, in percent. At k = 0 the low bound is 0 and at k = n the high bound is 100; the other bound is then the one-sided 95% bound, so the interval always contains the published value.",
  "k": "k = round(value x n). k_exact is false when no whole number of successes prints as the published value at its precision, usually an average over seeds or tasks. n_basis says whether the trial count is stated by the source, derived from its stated tasks x episodes, or counted by a site run.",
  "comparison": "Within one board, P(a > b) under independent Beta posteriors, computed exactly. A pair is separated when P(a > b) is above 1 - alpha/(2m) or below alpha/(2m), with alpha 0.05 and m pairs on the board.",
  "letters": "Compact letter display: rows that share a letter are not separated. Letters run a to z, then aa, ab and so on; on a board whose letter_groups exceeds 26 each row's letters are comma-separated.",
  "scope": "Rows on different boards are never compared. A row without a trial count keeps its published value and gets no interval or letter."
 },
 "boards": [
  {
   "id": "behavior-challenge-2025-held-out-test-50-tasks-privileged-track",
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "privileged track",
   "metric": "full-task success rate",
   "protocol": "10 held-out instances per task, 1 run each",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 1,
   "rows": [
    "behavior-challenge-2025-held-out-test-50-tasks-privileged-track-embodied-intelligence"
   ],
   "comparisons": []
  },
  {
   "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track",
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "standard track",
   "metric": "full-task success rate",
   "protocol": "10 held-out instances per task, 1 run each",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.00833333,
   "letter_groups": 1,
   "rows": [
    "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective",
    "behavior-challenge-2025-held-out-test-50-tasks-standard-track-comet-nvidia-research",
    "behavior-challenge-2025-held-out-test-50-tasks-standard-track-simpleai-robot",
    "behavior-challenge-2025-held-out-test-50-tasks-standard-track-the-north-star-huawei-cri-eai"
   ],
   "comparisons": [
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-comet-nvidia-research",
     "p_a_greater": 0.68662,
     "separated": false
    },
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-simpleai-robot",
     "p_a_greater": 0.7843,
     "separated": false
    },
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-the-north-star-huawei-cri-eai",
     "p_a_greater": 0.994257,
     "separated": false
    },
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-comet-nvidia-research",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-simpleai-robot",
     "p_a_greater": 0.618147,
     "separated": false
    },
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-comet-nvidia-research",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-the-north-star-huawei-cri-eai",
     "p_a_greater": 0.979521,
     "separated": false
    },
    {
     "a": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-simpleai-robot",
     "b": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-the-north-star-huawei-cri-eai",
     "p_a_greater": 0.959469,
     "separated": false
    }
   ]
  },
  {
   "id": "libero-plus-libero-all-four-suites-background-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Background perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-background-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-background-perturbation-pi0",
    "libero-plus-libero-all-four-suites-background-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-background-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-camera-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Camera perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-camera-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-camera-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-camera-perturbation-pi0",
    "libero-plus-libero-all-four-suites-camera-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-language-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Language perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-language-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-language-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-language-perturbation-pi0",
    "libero-plus-libero-all-four-suites-language-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-layout-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Layout perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-layout-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-layout-perturbation-pi0",
    "libero-plus-libero-all-four-suites-layout-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-layout-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-light-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Light perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-light-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-light-perturbation-pi0",
    "libero-plus-libero-all-four-suites-light-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-light-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-noise-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Noise perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-noise-perturbation-pi0",
    "libero-plus-libero-all-four-suites-noise-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-noise-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-noise-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-robot-perturbation",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Robot perturbation",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-robot-perturbation-openvla-oft",
    "libero-plus-libero-all-four-suites-robot-perturbation-pi0-fast",
    "libero-plus-libero-all-four-suites-robot-perturbation-pi0",
    "libero-plus-libero-all-four-suites-robot-perturbation-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-plus-libero-all-four-suites-in-distribution",
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "in-distribution",
   "metric": "success rate",
   "protocol": "LIBERO-Plus Table 1",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "libero-plus-libero-all-four-suites-in-distribution-openvla-oft",
    "libero-plus-libero-all-four-suites-in-distribution-pi0",
    "libero-plus-libero-all-four-suites-in-distribution-pi0-fast",
    "libero-plus-libero-all-four-suites-in-distribution-openvla"
   ],
   "comparisons": []
  },
  {
   "id": "libero-libero-goal-in-distribution",
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "metric": "success rate",
   "protocol": "OpenVLA-OFT Table I protocol",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 2,
   "rows": [
    "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1",
    "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
    "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
    "libero-libero-goal-in-distribution-pi0-fine-tuned",
    "libero-libero-goal-in-distribution-pi0-fast-fine-tuned",
    "libero-libero-goal-in-distribution-dit-policy-fine-tuned",
    "libero-libero-goal-in-distribution-octo-fine-tuned",
    "libero-libero-goal-in-distribution-openvla-fine-tuned",
    "libero-libero-goal-in-distribution-mdt-scratch-2-language-annotations",
    "libero-libero-goal-in-distribution-diffusion-policy-scratch"
   ],
   "comparisons": [
    {
     "a": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "p_a_greater": 0.953149,
     "separated": false
    },
    {
     "a": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.992574,
     "separated": true
    },
    {
     "a": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "b": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.779441,
     "separated": false
    }
   ]
  },
  {
   "id": "libero-libero-long-in-distribution",
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "metric": "success rate",
   "protocol": "OpenVLA-OFT Table I protocol",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 1,
   "rows": [
    "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1",
    "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
    "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
    "libero-libero-long-in-distribution-seer-pretrained-on-libero-90-fine-tuned",
    "libero-libero-long-in-distribution-pi0-fine-tuned",
    "libero-libero-long-in-distribution-seer-scratch",
    "libero-libero-long-in-distribution-mdt-scratch-2-language-annotations",
    "libero-libero-long-in-distribution-dit-policy-fine-tuned",
    "libero-libero-long-in-distribution-pi0-fast-fine-tuned",
    "libero-libero-long-in-distribution-openvla-fine-tuned",
    "libero-libero-long-in-distribution-octo-fine-tuned",
    "libero-libero-long-in-distribution-diffusion-policy-scratch"
   ],
   "comparisons": [
    {
     "a": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.782451,
     "separated": false
    },
    {
     "a": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "p_a_greater": 0.984943,
     "separated": false
    },
    {
     "a": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "b": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "p_a_greater": 0.918016,
     "separated": false
    }
   ]
  },
  {
   "id": "libero-libero-object-in-distribution",
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "metric": "success rate",
   "protocol": "OpenVLA-OFT Table I protocol",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 2,
   "rows": [
    "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1",
    "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
    "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
    "libero-libero-object-in-distribution-pi0-fine-tuned",
    "libero-libero-object-in-distribution-pi0-fast-fine-tuned",
    "libero-libero-object-in-distribution-dit-policy-fine-tuned",
    "libero-libero-object-in-distribution-diffusion-policy-scratch",
    "libero-libero-object-in-distribution-openvla-fine-tuned",
    "libero-libero-object-in-distribution-mdt-scratch-2-language-annotations",
    "libero-libero-object-in-distribution-octo-fine-tuned"
   ],
   "comparisons": [
    {
     "a": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "p_a_greater": 0.5,
     "separated": false
    },
    {
     "a": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.999815,
     "separated": true
    },
    {
     "a": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "b": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.999815,
     "separated": true
    }
   ]
  },
  {
   "id": "libero-libero-spatial-in-distribution",
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "metric": "success rate",
   "protocol": "OpenVLA-OFT Table I protocol",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 1,
   "rows": [
    "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1",
    "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
    "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
    "libero-libero-spatial-in-distribution-pi0-fine-tuned",
    "libero-libero-spatial-in-distribution-pi0-fast-fine-tuned",
    "libero-libero-spatial-in-distribution-openvla-fine-tuned",
    "libero-libero-spatial-in-distribution-dit-policy-fine-tuned",
    "libero-libero-spatial-in-distribution-octo-fine-tuned",
    "libero-libero-spatial-in-distribution-mdt-scratch-2-language-annotations",
    "libero-libero-spatial-in-distribution-diffusion-policy-scratch"
   ],
   "comparisons": [
    {
     "a": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "p_a_greater": 0.896052,
     "separated": false
    },
    {
     "a": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1",
     "b": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.978484,
     "separated": false
    },
    {
     "a": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
     "b": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
     "p_a_greater": 0.779441,
     "separated": false
    }
   ]
  },
  {
   "id": "libero-libero-spatial-site-bounded-trials",
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "site bounded trials",
   "metric": "success rate",
   "protocol": "site run, bounded executed trials",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 1,
   "rows": [
    "site.libero-spatial-task0-seed7"
   ],
   "comparisons": []
  },
  {
   "id": "maniskill3-pickcube-v1-reference-seeds",
   "benchmark": "ManiSkill3",
   "split": "PickCube-v1",
   "condition": "reference seeds",
   "metric": "success rate",
   "protocol": "site run, original 50-action horizon",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 1,
   "rows": [
    "site.maniskill3.pickcube.ppo"
   ],
   "comparisons": []
  },
  {
   "id": "robochallenge-table30-30-tasks-average-generalist-multi-task",
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "generalist multi-task",
   "metric": "success rate",
   "protocol": "10 rollouts per task",
   "medium": [
    "real"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 2,
   "rows": [
    "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-5-multi",
    "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-multi"
   ],
   "comparisons": [
    {
     "a": "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-5-multi",
     "b": "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-multi",
     "p_a_greater": 0.9986,
     "separated": true
    }
   ]
  },
  {
   "id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune",
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "task-specific fine-tune",
   "metric": "success rate",
   "protocol": "10 rollouts per task",
   "medium": [
    "real"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 3,
   "rows": [
    "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0-5",
    "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0",
    "robochallenge-table30-30-tasks-average-task-specific-fine-tune-cogact"
   ],
   "comparisons": [
    {
     "a": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0-5",
     "b": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0",
     "p_a_greater": 0.999955,
     "separated": true
    },
    {
     "a": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0-5",
     "b": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-cogact",
     "p_a_greater": 1.0,
     "separated": true
    },
    {
     "a": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0",
     "b": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-cogact",
     "p_a_greater": 1.0,
     "separated": true
    }
   ]
  },
  {
   "id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching",
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "SIMPLER visual matching",
   "metric": "success rate",
   "protocol": "SimplerEnv paper Table V",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-octo-small",
    "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-octo-base",
    "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-rt-1-x"
   ],
   "comparisons": []
  },
  {
   "id": "simplerenv-widowx-put-spoon-on-towel-real-robot",
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "real robot",
   "metric": "success rate",
   "protocol": "SimplerEnv paper Table V",
   "medium": [
    "real"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 2,
   "rows": [
    "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-small",
    "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-base",
    "simplerenv-widowx-put-spoon-on-towel-real-robot-rt-1-x"
   ],
   "comparisons": [
    {
     "a": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-small",
     "b": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-base",
     "p_a_greater": 0.719628,
     "separated": false
    },
    {
     "a": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-small",
     "b": "simplerenv-widowx-put-spoon-on-towel-real-robot-rt-1-x",
     "p_a_greater": 0.999881,
     "separated": true
    },
    {
     "a": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-base",
     "b": "simplerenv-widowx-put-spoon-on-towel-real-robot-rt-1-x",
     "p_a_greater": 0.999185,
     "separated": true
    }
   ]
  },
  {
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching",
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "SIMPLER visual matching",
   "metric": "success rate",
   "protocol": "SimplerEnv paper Table V",
   "medium": [
    "sim"
   ],
   "pairwise_threshold": 0.05,
   "letter_groups": 0,
   "rows": [
    "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-octo-small",
    "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-octo-base",
    "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-rt-1-x"
   ],
   "comparisons": []
  },
  {
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot",
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "real robot",
   "metric": "success rate",
   "protocol": "SimplerEnv paper Table V",
   "medium": [
    "real"
   ],
   "pairwise_threshold": 0.01666667,
   "letter_groups": 1,
   "rows": [
    "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-small",
    "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-base",
    "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-rt-1-x"
   ],
   "comparisons": [
    {
     "a": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-small",
     "b": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-base",
     "p_a_greater": 0.945072,
     "separated": false
    },
    {
     "a": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-small",
     "b": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-rt-1-x",
     "p_a_greater": 0.945072,
     "separated": false
    },
    {
     "a": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-base",
     "b": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-rt-1-x",
     "p_a_greater": 0.5,
     "separated": false
    }
   ]
  }
 ],
 "rows": [
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "privileged track",
   "policy": "Embodied Intelligence",
   "metric": "full-task success rate",
   "value_percent": 5.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "challenge",
   "reported_in": "original (held-out entries 'verified by the BEHAVIOR team')",
   "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
   "locator": "Leaderboard table, Held-out Test columns",
   "quote": "Embodied Intelligence ... 0.0520",
   "protocol": "10 held-out instances per task, 1 run each",
   "n_basis": "derived",
   "note": "Rules (archive/2025/evaluation.html): 'Averaged across 50 tasks', 'hold out 10 more instances for final evaluation', 'evaluate your policy 1 time' per instance. 500 = 50×10×1 is derived; neither page states a total. 1st-place report (arXiv 2512.06951 §8.1) confirms 'held-out set of 10 instances per task'.",
   "id": "behavior-challenge-2025-held-out-test-50-tasks-privileged-track-embodied-intelligence",
   "k": 26,
   "k_exact": true,
   "posterior_mean": 5.38,
   "interval_low": 3.58,
   "interval_high": 7.51,
   "interval_width": 3.93,
   "letters": "a",
   "board": "behavior-challenge-2025-held-out-test-50-tasks-privileged-track"
  },
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "standard track",
   "policy": "Robot Learning Collective",
   "metric": "full-task success rate",
   "value_percent": 12.4,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "challenge",
   "reported_in": "original (held-out entries 'verified by the BEHAVIOR team')",
   "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
   "locator": "Leaderboard table, Held-out Test columns",
   "quote": "Robot Learning Collective ... 0.1240",
   "protocol": "10 held-out instances per task, 1 run each",
   "n_basis": "derived",
   "note": "Rules (archive/2025/evaluation.html): 'Averaged across 50 tasks', 'hold out 10 more instances for final evaluation', 'evaluate your policy 1 time' per instance. 500 = 50×10×1 is derived; neither page states a total. 1st-place report (arXiv 2512.06951 §8.1) confirms 'held-out set of 10 instances per task'.",
   "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective",
   "k": 62,
   "k_exact": true,
   "posterior_mean": 12.55,
   "interval_low": 9.8,
   "interval_high": 15.58,
   "interval_width": 5.78,
   "letters": "a",
   "board": "behavior-challenge-2025-held-out-test-50-tasks-standard-track"
  },
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "standard track",
   "policy": "Comet (NVIDIA Research)",
   "metric": "full-task success rate",
   "value_percent": 11.4,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "challenge",
   "reported_in": "original (held-out entries 'verified by the BEHAVIOR team')",
   "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
   "locator": "Leaderboard table, Held-out Test columns",
   "quote": "Comet ... 0.1140",
   "protocol": "10 held-out instances per task, 1 run each",
   "n_basis": "derived",
   "note": "Rules (archive/2025/evaluation.html): 'Averaged across 50 tasks', 'hold out 10 more instances for final evaluation', 'evaluate your policy 1 time' per instance. 500 = 50×10×1 is derived; neither page states a total. 1st-place report (arXiv 2512.06951 §8.1) confirms 'held-out set of 10 instances per task'.",
   "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-comet-nvidia-research",
   "k": 57,
   "k_exact": true,
   "posterior_mean": 11.55,
   "interval_low": 8.91,
   "interval_high": 14.49,
   "interval_width": 5.58,
   "letters": "a",
   "board": "behavior-challenge-2025-held-out-test-50-tasks-standard-track"
  },
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "standard track",
   "policy": "SimpleAI Robot",
   "metric": "full-task success rate",
   "value_percent": 10.8,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "challenge",
   "reported_in": "original (held-out entries 'verified by the BEHAVIOR team')",
   "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
   "locator": "Leaderboard table, Held-out Test columns",
   "quote": "SimpleAI Robot ... 0.1080",
   "protocol": "10 held-out instances per task, 1 run each",
   "n_basis": "derived",
   "note": "Rules (archive/2025/evaluation.html): 'Averaged across 50 tasks', 'hold out 10 more instances for final evaluation', 'evaluate your policy 1 time' per instance. 500 = 50×10×1 is derived; neither page states a total. 1st-place report (arXiv 2512.06951 §8.1) confirms 'held-out set of 10 instances per task'.",
   "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-simpleai-robot",
   "k": 54,
   "k_exact": true,
   "posterior_mean": 10.96,
   "interval_low": 8.38,
   "interval_high": 13.83,
   "interval_width": 5.45,
   "letters": "a",
   "board": "behavior-challenge-2025-held-out-test-50-tasks-standard-track"
  },
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "split": "held-out test, 50 tasks",
   "condition": "standard track",
   "policy": "The North Star (Huawei CRI EAI)",
   "metric": "full-task success rate",
   "value_percent": 7.6,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "challenge",
   "reported_in": "original (held-out entries 'verified by the BEHAVIOR team')",
   "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
   "locator": "Leaderboard table, Held-out Test columns",
   "quote": "The North Star ... 0.0760",
   "protocol": "10 held-out instances per task, 1 run each",
   "n_basis": "derived",
   "note": "Rules (archive/2025/evaluation.html): 'Averaged across 50 tasks', 'hold out 10 more instances for final evaluation', 'evaluate your policy 1 time' per instance. 500 = 50×10×1 is derived; neither page states a total. 1st-place report (arXiv 2512.06951 §8.1) confirms 'held-out set of 10 instances per task'.",
   "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-the-north-star-huawei-cri-eai",
   "k": 38,
   "k_exact": true,
   "posterior_mean": 7.77,
   "interval_low": 5.59,
   "interval_high": 10.26,
   "interval_width": 4.67,
   "letters": "a",
   "board": "behavior-challenge-2025-held-out-test-50-tasks-standard-track"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
   "metric": "success rate",
   "value_percent": 97.9,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1): 97.6 | 98.4 | 97.9 | 94.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1",
   "k": 490,
   "k_exact": false,
   "posterior_mean": 97.81,
   "interval_low": 96.36,
   "interval_high": 98.9,
   "interval_width": 2.54,
   "letters": "a",
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
   "metric": "success rate",
   "value_percent": 96.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only: 96.2 | 98.3 | 96.2 | 90.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
   "k": 481,
   "k_exact": true,
   "posterior_mean": 96.02,
   "interval_low": 94.14,
   "interval_high": 97.54,
   "interval_width": 3.4,
   "letters": "ab",
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset",
   "metric": "success rate",
   "value_percent": 95.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset: 95.2 | 94.2 | 95.2 | 93.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
   "k": 476,
   "k_exact": true,
   "posterior_mean": 95.02,
   "interval_low": 92.96,
   "interval_high": 96.75,
   "interval_width": 3.79,
   "letters": "b",
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "π0 (fine-tuned)",
   "metric": "success rate",
   "value_percent": 95.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper (cited [3]), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0 (fine-tuned): 96.8 | 98.8 | 95.8 | 85.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-pi0-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "π0-FAST (fine-tuned)",
   "metric": "success rate",
   "value_percent": 88.6,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [38], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0-FAST (fine-tuned): 96.4 | 96.8 | 88.6 | 60.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-pi0-fast-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "DiT Policy (fine-tuned)",
   "metric": "success rate",
   "value_percent": 85.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [13], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "DiT Policy (fine-tuned): 84.2 | 96.3 | 85.4 | 63.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-dit-policy-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "Octo (fine-tuned)",
   "metric": "success rate",
   "value_percent": 84.6,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Octo (fine-tuned): 78.9 | 85.7 | 84.6 | 51.1 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-octo-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "OpenVLA (fine-tuned)",
   "metric": "success rate",
   "value_percent": 79.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA (fine-tuned): 84.7 | 88.4 | 79.2 | 53.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-openvla-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "MDT (scratch; 2% language annotations)",
   "metric": "success rate",
   "value_percent": 73.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [40], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "MDT (scratch; 2% language annotations): 78.5 | 87.5 | 73.5 | 64.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-goal-in-distribution-mdt-scratch-2-language-annotations",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Goal",
   "condition": "in-distribution",
   "policy": "Diffusion Policy (scratch)",
   "metric": "success rate",
   "value_percent": 68.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Diffusion Policy (scratch): 78.3 | 92.5 | 68.3 | 50.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-goal-in-distribution-diffusion-policy-scratch",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-goal-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
   "metric": "success rate",
   "value_percent": 94.5,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1): 97.6 | 98.4 | 97.9 | 94.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1",
   "k": 472,
   "k_exact": false,
   "posterior_mean": 94.22,
   "interval_low": 92.02,
   "interval_high": 96.09,
   "interval_width": 4.07,
   "letters": "a",
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset",
   "metric": "success rate",
   "value_percent": 93.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset: 95.2 | 94.2 | 95.2 | 93.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
   "k": 466,
   "k_exact": true,
   "posterior_mean": 93.03,
   "interval_low": 90.65,
   "interval_high": 95.09,
   "interval_width": 4.44,
   "letters": "a",
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
   "metric": "success rate",
   "value_percent": 90.7,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only: 96.2 | 98.3 | 96.2 | 90.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
   "k": 454,
   "k_exact": false,
   "posterior_mean": 90.64,
   "interval_low": 87.94,
   "interval_high": 93.03,
   "interval_width": 5.08,
   "letters": "a",
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "Seer (pretrained on LIBERO-90, fine-tuned)",
   "metric": "success rate",
   "value_percent": 87.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [50], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Seer (pretrained on LIBERO-90, fine-tuned): – | – | – | 87.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-long-in-distribution-seer-pretrained-on-libero-90-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "π0 (fine-tuned)",
   "metric": "success rate",
   "value_percent": 85.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper (cited [3]), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0 (fine-tuned): 96.8 | 98.8 | 95.8 | 85.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-long-in-distribution-pi0-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "Seer (scratch)",
   "metric": "success rate",
   "value_percent": 78.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [50], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Seer (scratch): – | – | – | 78.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-long-in-distribution-seer-scratch",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "MDT (scratch; 2% language annotations)",
   "metric": "success rate",
   "value_percent": 64.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [40], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "MDT (scratch; 2% language annotations): 78.5 | 87.5 | 73.5 | 64.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-long-in-distribution-mdt-scratch-2-language-annotations",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "DiT Policy (fine-tuned)",
   "metric": "success rate",
   "value_percent": 63.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [13], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "DiT Policy (fine-tuned): 84.2 | 96.3 | 85.4 | 63.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-long-in-distribution-dit-policy-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "π0-FAST (fine-tuned)",
   "metric": "success rate",
   "value_percent": 60.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [38], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0-FAST (fine-tuned): 96.4 | 96.8 | 88.6 | 60.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-long-in-distribution-pi0-fast-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "OpenVLA (fine-tuned)",
   "metric": "success rate",
   "value_percent": 53.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA (fine-tuned): 84.7 | 88.4 | 79.2 | 53.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-long-in-distribution-openvla-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "Octo (fine-tuned)",
   "metric": "success rate",
   "value_percent": 51.1,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Octo (fine-tuned): 78.9 | 85.7 | 84.6 | 51.1 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-long-in-distribution-octo-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Long",
   "condition": "in-distribution",
   "policy": "Diffusion Policy (scratch)",
   "metric": "success rate",
   "value_percent": 50.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Diffusion Policy (scratch): 78.3 | 92.5 | 68.3 | 50.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-long-in-distribution-diffusion-policy-scratch",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-long-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
   "metric": "success rate",
   "value_percent": 98.4,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1): 97.6 | 98.4 | 97.9 | 94.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1",
   "k": 492,
   "k_exact": true,
   "posterior_mean": 98.21,
   "interval_low": 96.88,
   "interval_high": 99.18,
   "interval_width": 2.3,
   "letters": "a",
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
   "metric": "success rate",
   "value_percent": 98.3,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only: 96.2 | 98.3 | 96.2 | 90.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
   "k": 492,
   "k_exact": false,
   "posterior_mean": 98.21,
   "interval_low": 96.88,
   "interval_high": 99.18,
   "interval_width": 2.3,
   "letters": "a",
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset",
   "metric": "success rate",
   "value_percent": 94.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset: 95.2 | 94.2 | 95.2 | 93.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-object-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
   "k": 471,
   "k_exact": true,
   "posterior_mean": 94.02,
   "interval_low": 91.79,
   "interval_high": 95.92,
   "interval_width": 4.13,
   "letters": "b",
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "π0 (fine-tuned)",
   "metric": "success rate",
   "value_percent": 98.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper (cited [3]), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0 (fine-tuned): 96.8 | 98.8 | 95.8 | 85.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-object-in-distribution-pi0-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "π0-FAST (fine-tuned)",
   "metric": "success rate",
   "value_percent": 96.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [38], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0-FAST (fine-tuned): 96.4 | 96.8 | 88.6 | 60.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-object-in-distribution-pi0-fast-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "DiT Policy (fine-tuned)",
   "metric": "success rate",
   "value_percent": 96.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [13], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "DiT Policy (fine-tuned): 84.2 | 96.3 | 85.4 | 63.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-object-in-distribution-dit-policy-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "Diffusion Policy (scratch)",
   "metric": "success rate",
   "value_percent": 92.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Diffusion Policy (scratch): 78.3 | 92.5 | 68.3 | 50.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-object-in-distribution-diffusion-policy-scratch",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "OpenVLA (fine-tuned)",
   "metric": "success rate",
   "value_percent": 88.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA (fine-tuned): 84.7 | 88.4 | 79.2 | 53.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-object-in-distribution-openvla-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "MDT (scratch; 2% language annotations)",
   "metric": "success rate",
   "value_percent": 87.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [40], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "MDT (scratch; 2% language annotations): 78.5 | 87.5 | 73.5 | 64.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-object-in-distribution-mdt-scratch-2-language-annotations",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Object",
   "condition": "in-distribution",
   "policy": "Octo (fine-tuned)",
   "metric": "success rate",
   "value_percent": 85.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Octo (fine-tuned): 78.9 | 85.7 | 84.6 | 51.1 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-object-in-distribution-octo-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-object-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
   "metric": "success rate",
   "value_percent": 97.6,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1): 97.6 | 98.4 | 97.9 | 94.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1",
   "k": 488,
   "k_exact": true,
   "posterior_mean": 97.41,
   "interval_low": 95.85,
   "interval_high": 98.61,
   "interval_width": 2.76,
   "letters": "a",
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
   "metric": "success rate",
   "value_percent": 96.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only: 96.2 | 98.3 | 96.2 | 90.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-third-person-image-only",
   "k": 481,
   "k_exact": true,
   "posterior_mean": 96.02,
   "interval_low": 94.14,
   "interval_high": 97.54,
   "interval_width": 3.4,
   "letters": "a",
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset",
   "metric": "success rate",
   "value_percent": 95.2,
   "n_trials": 500,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset: 95.2 | 94.2 | 95.2 | 93.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": "stated",
   "note": "'OpenVLA results from this work are averaged over 500 trials for each task suite (10 tasks × 50 episodes)' (Table I caption). Seeds not stated; best checkpoint per run reported (Sec. V-A). Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1-unfiltered-dataset",
   "k": 476,
   "k_exact": true,
   "posterior_mean": 95.02,
   "interval_low": 92.96,
   "interval_high": 96.75,
   "interval_width": 3.79,
   "letters": "a",
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "π0 (fine-tuned)",
   "metric": "success rate",
   "value_percent": 96.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper (cited [3]), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0 (fine-tuned): 96.8 | 98.8 | 95.8 | 85.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-pi0-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "π0-FAST (fine-tuned)",
   "metric": "success rate",
   "value_percent": 96.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [38], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "π0-FAST (fine-tuned): 96.4 | 96.8 | 88.6 | 60.2 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-pi0-fast-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "OpenVLA (fine-tuned)",
   "metric": "success rate",
   "value_percent": 84.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "OpenVLA (fine-tuned): 84.7 | 88.4 | 79.2 | 53.7 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-openvla-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "DiT Policy (fine-tuned)",
   "metric": "success rate",
   "value_percent": 84.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [13], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "DiT Policy (fine-tuned): 84.2 | 96.3 | 85.4 | 63.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-dit-policy-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "Octo (fine-tuned)",
   "metric": "success rate",
   "value_percent": 78.9,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Octo (fine-tuned): 78.9 | 85.7 | 84.6 | 51.1 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-octo-fine-tuned",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "MDT (scratch; 2% language annotations)",
   "metric": "success rate",
   "value_percent": 78.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "self",
   "reported_in": "copied from original paper [40], per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "MDT (scratch; 2% language annotations): 78.5 | 87.5 | 73.5 | 64.8 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person + wrist + proprio; original unfiltered dataset.",
   "id": "libero-libero-spatial-in-distribution-mdt-scratch-2-language-annotations",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "in-distribution",
   "policy": "Diffusion Policy (scratch)",
   "metric": "success rate",
   "value_percent": 78.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "copied from Kim et al. [23] (OpenVLA paper), per Table I caption",
   "source_url": "https://arxiv.org/abs/2502.19645",
   "locator": "Table I (PDF)",
   "quote": "Diffusion Policy (scratch): 78.3 | 92.5 | 68.3 | 50.5 (Spatial|Object|Goal|Long)",
   "protocol": "OpenVLA-OFT Table I protocol",
   "n_basis": null,
   "note": "Baseline copied from another paper; Table I's 500-trial statement covers only 'OpenVLA results from this work'. Check original paper for n. Input group: third-person image + language; filtered dataset.",
   "id": "libero-libero-spatial-in-distribution-diffusion-policy-scratch",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-libero-spatial-in-distribution"
  },
  {
   "id": "site.libero-spatial-task0-seed7",
   "benchmark": "LIBERO",
   "split": "LIBERO-Spatial",
   "condition": "site bounded trials",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "protocol": "site run, bounded executed trials",
   "value_percent": 100.0,
   "n_trials": 1,
   "n_basis": "site",
   "seeds": null,
   "medium": "sim",
   "evaluator": "site",
   "source_url": "/evidence/native-evaluation/libero-spatial-task0-seed7/result.json",
   "locator": "result.json",
   "note": "Bounded executed trials only; not the 500-episode paper protocol.",
   "k": 1,
   "k_exact": true,
   "posterior_mean": 66.67,
   "interval_low": 22.36,
   "interval_high": 100.0,
   "interval_width": 77.64,
   "letters": "a",
   "board": "libero-libero-spatial-site-bounded-trials"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Background perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 92.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-background-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-background-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Background perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 78.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-background-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-background-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Background perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 67.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-background-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-background-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Background perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 25.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-background-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-background-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Camera perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 66.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-camera-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-camera-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Camera perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 59.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-camera-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-camera-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Camera perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 15.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-camera-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-camera-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Camera perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 1.1,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-camera-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-camera-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "in-distribution",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 97.1,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "unclear (matches 2502.19645 Table I)",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "'Original' value equals OpenVLA-OFT Table I exactly; paper does not say if re-run or copied. Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-in-distribution-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-in-distribution"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "in-distribution",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 94.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "unclear (matches 2502.19645 Table I)",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "'Original' value equals OpenVLA-OFT Table I exactly; paper does not say if re-run or copied. Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-in-distribution-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-in-distribution"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "in-distribution",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 85.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "unclear (matches 2502.19645 Table I)",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "'Original' value equals OpenVLA-OFT Table I exactly; paper does not say if re-run or copied. Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-in-distribution-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-in-distribution"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "in-distribution",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 76.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "unclear (matches 2502.19645 Table I)",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "'Original' value equals OpenVLA-OFT Table I exactly; paper does not say if re-run or copied. Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-in-distribution-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-in-distribution"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Language perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 81.5,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-language-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-language-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Language perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 63.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-language-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-language-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Language perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 61.0,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-language-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-language-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Language perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 26.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-language-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-language-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Layout perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 77.1,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-layout-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-layout-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Layout perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 70.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-layout-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-layout-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Layout perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 70.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-layout-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-layout-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Layout perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 31.6,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-layout-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-layout-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Light perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 85.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-light-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-light-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Light perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 79.6,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-light-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-light-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Light perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 73.0,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-light-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-light-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Light perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 4.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-light-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-light-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Noise perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 79.4,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-noise-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-noise-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Noise perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 76.7,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-noise-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-noise-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Noise perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 75.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-noise-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-noise-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Noise perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 19.3,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-noise-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-noise-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Robot perturbation",
   "policy": "OpenVLA-OFT",
   "metric": "success rate",
   "value_percent": 37.2,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA-OFT: 97.1 | 59.7 | 37.2 | 81.5 | 85.8 | 92.4 | 76.7 | 77.1",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-robot-perturbation-openvla-oft",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-robot-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Robot perturbation",
   "policy": "π0-FAST",
   "metric": "success rate",
   "value_percent": 24.8,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0-FAST: 85.5 | 66.4 | 24.8 | 63.3 | 73.0 | 67.7 | 75.8 | 70.3",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-robot-perturbation-pi0-fast",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-robot-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Robot perturbation",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 6.6,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "π0: 94.2 | 15.8 | 6.6 | 61.0 | 79.6 | 78.5 | 79.4 | 70.4",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-robot-perturbation-pi0",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-robot-perturbation"
  },
  {
   "benchmark": "LIBERO-Plus",
   "split": "LIBERO (all four suites)",
   "condition": "Robot perturbation",
   "policy": "OpenVLA",
   "metric": "success rate",
   "value_percent": 4.1,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.13626",
   "locator": "Table 1 'Model performance under different perturbations' (HTML v3)",
   "quote": "OpenVLA: 76.5 | 1.1 | 4.1 | 26.8 | 4.4 | 25.3 | 19.3 | 31.6",
   "protocol": "LIBERO-Plus Table 1",
   "n_basis": null,
   "note": "Paper text states no trial/episode count per perturbation; GitHub README says 10,030 tasks and default num_trials_per_task 50 but does not tie that to Table 1. README leaderboard shows different (newer/older) numbers than paper v3 Table 1.",
   "id": "libero-plus-libero-all-four-suites-robot-perturbation-openvla",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "libero-plus-libero-all-four-suites-robot-perturbation"
  },
  {
   "id": "site.maniskill3.pickcube.ppo",
   "benchmark": "ManiSkill3",
   "split": "PickCube-v1",
   "condition": "reference seeds",
   "policy": "ManiSkill PPO reference checkpoint",
   "metric": "success rate",
   "protocol": "site run, original 50-action horizon",
   "value_percent": 100.0,
   "n_trials": 10,
   "n_basis": "site",
   "seeds": null,
   "medium": "sim",
   "evaluator": "site",
   "source_url": "/worlds/maniskill-sources.html",
   "locator": "worlds/maniskill.json results",
   "note": "PickCube-v1 only; one released PPO checkpoint; not the full suite or a paper score",
   "k": 10,
   "k_exact": true,
   "posterior_mean": 91.67,
   "interval_low": 76.16,
   "interval_high": 100.0,
   "interval_width": 23.84,
   "letters": "a",
   "board": "maniskill3-pickcube-v1-reference-seeds"
  },
  {
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "generalist multi-task",
   "policy": "π0.5/multi",
   "metric": "success rate",
   "value_percent": 17.7,
   "n_trials": 300,
   "seeds": null,
   "medium": "real",
   "evaluator": "challenge",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.17950",
   "locator": "Figure 9 (rendered as table), 'average' row, v1",
   "quote": "average | 43.7 | 62.2 | 28.3 | 47.6 | 11.7 | 21.8 | 17.7 | 31.3 | 9.3 | 20.6",
   "protocol": "10 rollouts per task",
   "n_basis": "derived",
   "note": "'We make 10 rollouts for each task.' + 'initial release of the tasks includes 30 tasks'. 300 = 30×10 is derived; paper states no total. Real robots; models run by submitter, robots served by remote API.",
   "id": "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-5-multi",
   "k": 53,
   "k_exact": true,
   "posterior_mean": 17.88,
   "interval_low": 13.77,
   "interval_high": 22.39,
   "interval_width": 8.62,
   "letters": "a",
   "board": "robochallenge-table30-30-tasks-average-generalist-multi-task"
  },
  {
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "generalist multi-task",
   "policy": "π0/multi",
   "metric": "success rate",
   "value_percent": 9.3,
   "n_trials": 300,
   "seeds": null,
   "medium": "real",
   "evaluator": "challenge",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.17950",
   "locator": "Figure 9 (rendered as table), 'average' row, v1",
   "quote": "average | 43.7 | 62.2 | 28.3 | 47.6 | 11.7 | 21.8 | 17.7 | 31.3 | 9.3 | 20.6",
   "protocol": "10 rollouts per task",
   "n_basis": "derived",
   "note": "'We make 10 rollouts for each task.' + 'initial release of the tasks includes 30 tasks'. 300 = 30×10 is derived; paper states no total. Real robots; models run by submitter, robots served by remote API.",
   "id": "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-multi",
   "k": 28,
   "k_exact": true,
   "posterior_mean": 9.6,
   "interval_low": 6.55,
   "interval_high": 13.16,
   "interval_width": 6.62,
   "letters": "b",
   "board": "robochallenge-table30-30-tasks-average-generalist-multi-task"
  },
  {
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "task-specific fine-tune",
   "policy": "π0.5",
   "metric": "success rate",
   "value_percent": 43.7,
   "n_trials": 300,
   "seeds": null,
   "medium": "real",
   "evaluator": "challenge",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.17950",
   "locator": "Figure 9 (rendered as table), 'average' row, v1",
   "quote": "average | 43.7 | 62.2 | 28.3 | 47.6 | 11.7 | 21.8 | 17.7 | 31.3 | 9.3 | 20.6",
   "protocol": "10 rollouts per task",
   "n_basis": "derived",
   "note": "'We make 10 rollouts for each task.' + 'initial release of the tasks includes 30 tasks'. 300 = 30×10 is derived; paper states no total. Real robots; models run by submitter, robots served by remote API.",
   "id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0-5",
   "k": 131,
   "k_exact": true,
   "posterior_mean": 43.71,
   "interval_low": 38.17,
   "interval_high": 49.33,
   "interval_width": 11.16,
   "letters": "a",
   "board": "robochallenge-table30-30-tasks-average-task-specific-fine-tune"
  },
  {
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "task-specific fine-tune",
   "policy": "π0",
   "metric": "success rate",
   "value_percent": 28.3,
   "n_trials": 300,
   "seeds": null,
   "medium": "real",
   "evaluator": "challenge",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.17950",
   "locator": "Figure 9 (rendered as table), 'average' row, v1",
   "quote": "average | 43.7 | 62.2 | 28.3 | 47.6 | 11.7 | 21.8 | 17.7 | 31.3 | 9.3 | 20.6",
   "protocol": "10 rollouts per task",
   "n_basis": "derived",
   "note": "'We make 10 rollouts for each task.' + 'initial release of the tasks includes 30 tasks'. 300 = 30×10 is derived; paper states no total. Real robots; models run by submitter, robots served by remote API.",
   "id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0",
   "k": 85,
   "k_exact": true,
   "posterior_mean": 28.48,
   "interval_low": 23.54,
   "interval_high": 33.69,
   "interval_width": 10.15,
   "letters": "b",
   "board": "robochallenge-table30-30-tasks-average-task-specific-fine-tune"
  },
  {
   "benchmark": "RoboChallenge Table30",
   "split": "30 tasks, average",
   "condition": "task-specific fine-tune",
   "policy": "CogACT",
   "metric": "success rate",
   "value_percent": 11.7,
   "n_trials": 300,
   "seeds": null,
   "medium": "real",
   "evaluator": "challenge",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2510.17950",
   "locator": "Figure 9 (rendered as table), 'average' row, v1",
   "quote": "average | 43.7 | 62.2 | 28.3 | 47.6 | 11.7 | 21.8 | 17.7 | 31.3 | 9.3 | 20.6",
   "protocol": "10 rollouts per task",
   "n_basis": "derived",
   "note": "'We make 10 rollouts for each task.' + 'initial release of the tasks includes 30 tasks'. 300 = 30×10 is derived; paper states no total. Real robots; models run by submitter, robots served by remote API.",
   "id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-cogact",
   "k": 35,
   "k_exact": true,
   "posterior_mean": 11.92,
   "interval_low": 8.52,
   "interval_high": 15.8,
   "interval_width": 7.28,
   "letters": "c",
   "board": "robochallenge-table30-30-tasks-average-task-specific-fine-tune"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "real robot",
   "policy": "Octo-Small",
   "metric": "success rate",
   "value_percent": 41.7,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval Octo-Small success 0.417",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-small",
   "k": 10,
   "k_exact": true,
   "posterior_mean": 42.31,
   "interval_low": 24.4,
   "interval_high": 61.33,
   "interval_width": 36.93,
   "letters": "a",
   "board": "simplerenv-widowx-put-spoon-on-towel-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "real robot",
   "policy": "Octo-Base",
   "metric": "success rate",
   "value_percent": 33.3,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval Octo-Base success 0.333",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-base",
   "k": 8,
   "k_exact": true,
   "posterior_mean": 34.62,
   "interval_low": 17.97,
   "interval_high": 53.5,
   "interval_width": 35.53,
   "letters": "a",
   "board": "simplerenv-widowx-put-spoon-on-towel-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "real robot",
   "policy": "RT-1-X",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval RT-1-X success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-put-spoon-on-towel-real-robot-rt-1-x",
   "k": 0,
   "k_exact": true,
   "posterior_mean": 3.85,
   "interval_low": 0.0,
   "interval_high": 11.29,
   "interval_width": 11.29,
   "letters": "b",
   "board": "simplerenv-widowx-put-spoon-on-towel-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "SIMPLER visual matching",
   "policy": "Octo-Small",
   "metric": "success rate",
   "value_percent": 47.2,
   "n_trials": null,
   "seeds": 3,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) Octo-Small success 0.472",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-octo-small",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "SIMPLER visual matching",
   "policy": "Octo-Base",
   "metric": "success rate",
   "value_percent": 12.5,
   "n_trials": null,
   "seeds": 3,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) Octo-Base success 0.125",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-octo-base",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Put Spoon on Towel",
   "condition": "SIMPLER visual matching",
   "policy": "RT-1-X",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) RT-1-X success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-rt-1-x",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "real robot",
   "policy": "Octo-Small",
   "metric": "success rate",
   "value_percent": 12.5,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval Octo-Small success 0.125",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-small",
   "k": 3,
   "k_exact": true,
   "posterior_mean": 15.38,
   "interval_low": 4.54,
   "interval_high": 31.22,
   "interval_width": 26.68,
   "letters": "a",
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "real robot",
   "policy": "Octo-Base",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval Octo-Base success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-base",
   "k": 0,
   "k_exact": true,
   "posterior_mean": 3.85,
   "interval_low": 0.0,
   "interval_high": 11.29,
   "interval_width": 11.29,
   "letters": "a",
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "real robot",
   "policy": "RT-1-X",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": 24,
   "seeds": null,
   "medium": "real",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (Real Eval) + Appendix B",
   "quote": "Real Eval RT-1-X success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": "stated",
   "note": "Appendix B: 'This creates a total of 2 × 12 = 24 trials.' Values are k/24 multiples.",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-rt-1-x",
   "k": 0,
   "k_exact": true,
   "posterior_mean": 3.85,
   "interval_low": 0.0,
   "interval_high": 11.29,
   "interval_width": 11.29,
   "letters": "a",
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "SIMPLER visual matching",
   "policy": "Octo-Small",
   "metric": "success rate",
   "value_percent": 4.2,
   "n_trials": null,
   "seeds": 3,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) Octo-Small success 0.042",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-octo-small",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "SIMPLER visual matching",
   "policy": "Octo-Base",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": null,
   "seeds": 3,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) Octo-Base success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-octo-base",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching"
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "split": "Stack Green Block on Yellow Block",
   "condition": "SIMPLER visual matching",
   "policy": "RT-1-X",
   "metric": "success rate",
   "value_percent": 0.0,
   "n_trials": null,
   "seeds": null,
   "medium": "sim",
   "evaluator": "third-party",
   "reported_in": "original",
   "source_url": "https://arxiv.org/abs/2405.05941",
   "locator": "Table V (SIMPLER Eval, Visual Matching)",
   "quote": "SIMPLER Eval (Visual Matching) RT-1-X success 0.000",
   "protocol": "SimplerEnv paper Table V",
   "n_basis": null,
   "note": "Appendix B: sim trials = real setup multiplied by tuned arm colors (Google Robot) 'along with the number of seeds for the Octo policies'; Sec VI-A: Octo averaged over three seeds. WidowX sim total not stated explicitly (Octo values are consistent with 72 = 24×3).",
   "id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-rt-1-x",
   "k": null,
   "k_exact": null,
   "posterior_mean": null,
   "interval_low": null,
   "interval_high": null,
   "interval_width": null,
   "letters": null,
   "board": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching"
  }
 ]
}
