{
 "$schema": "https://botlandscape.com/benchmark-health.schema.json",
 "schema_version": 1,
 "source": "catalogue/result-intervals.json",
 "min_verdict_policies": 5,
 "method": {
  "saturated": "a clean (unperturbed, simulated) board with at least 2 policies has a best published value of at least 95%",
  "open": "no board's best published value reaches 50%",
  "collapses": "the mean drop from in-distribution to some perturbation board with the same split, metric and protocol is at least 20 points, over at least 5 policies on both boards",
  "too-few-perturbed": "perturbation boards share policies with the in-distribution board, but none has the 5 needed for a verdict; the drops are shown and no collapses badge is given",
  "tracks-real": "sim and real boards are paired only when split, metric and protocol match. Every pair with at least 5 shared policies whose real board separates at least two of them has Spearman ρ ≥ 0.8 and MMRV ≤ 10 points; otherwise sim disagrees with real. Pairs whose real board separates nobody are shown but do not decide",
  "too-few-policies": "sim and real boards share policies, but no pair has the 5 needed for a verdict; the numbers are shown and no tracks or disagrees badge is given",
  "leaders-tied": "more than one policy shares a letter with the top row on a board with trial counts",
  "n-missing": "fewer than 50% of the benchmark's rows publish a trial count",
  "status": "the first badge present in this order: saturated, disagrees-real, collapses, open, tracks-real, real-robot, unvalidated"
 },
 "badges": [
  {
   "id": "saturated",
   "label": "Saturated"
  },
  {
   "id": "open",
   "label": "Open"
  },
  {
   "id": "collapses",
   "label": "Collapses under perturbation"
  },
  {
   "id": "too-few-perturbed",
   "label": "Too few policies to judge robustness"
  },
  {
   "id": "tracks-real",
   "label": "Sim tracks real"
  },
  {
   "id": "disagrees-real",
   "label": "Sim disagrees with real"
  },
  {
   "id": "too-few-policies",
   "label": "Too few policies to judge sim vs real"
  },
  {
   "id": "real-robot",
   "label": "Real robot results"
  },
  {
   "id": "unvalidated",
   "label": "Unvalidated vs real"
  },
  {
   "id": "leaders-tied",
   "label": "Leaders can't be separated"
  },
  {
   "id": "n-missing",
   "label": "Trial counts mostly missing"
  }
 ],
 "cards": [
  {
   "benchmark": "LIBERO",
   "status": "saturated",
   "badges": [
    "saturated",
    "unvalidated",
    "leaders-tied",
    "n-missing"
   ],
   "catalogue": {
    "id": "sim--sim-manipulation-eval--libero",
    "name": "LIBERO",
    "first_public": "arXiv 2023-06-05 [R5]",
    "href": "/topics/benchmarks.html#sim--sim-manipulation-eval"
   },
   "boards": [
    {
     "id": "libero-libero-goal-in-distribution",
     "split": "LIBERO-Goal",
     "condition": "in-distribution",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 10,
     "rows_with_n": 3,
     "best": {
      "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
      "value_percent": 97.9,
      "interval_low": 96.36,
      "interval_high": 98.9,
      "n_trials": 500,
      "row_id": "libero-libero-goal-in-distribution-openvla-oft-pd-ac-cont-l1"
     },
     "headroom": 2.1,
     "leaders": [
      "OpenVLA-OFT (PD&AC, Cont-L1)",
      "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only"
     ]
    },
    {
     "id": "libero-libero-long-in-distribution",
     "split": "LIBERO-Long",
     "condition": "in-distribution",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 12,
     "rows_with_n": 3,
     "best": {
      "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
      "value_percent": 94.5,
      "interval_low": 92.02,
      "interval_high": 96.09,
      "n_trials": 500,
      "row_id": "libero-libero-long-in-distribution-openvla-oft-pd-ac-cont-l1"
     },
     "headroom": 5.5,
     "leaders": [
      "OpenVLA-OFT (PD&AC, Cont-L1)",
      "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
      "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset"
     ]
    },
    {
     "id": "libero-libero-object-in-distribution",
     "split": "LIBERO-Object",
     "condition": "in-distribution",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 10,
     "rows_with_n": 3,
     "best": {
      "policy": "π0 (fine-tuned)",
      "value_percent": 98.8,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-libero-object-in-distribution-pi0-fine-tuned"
     },
     "headroom": 1.2,
     "leaders": [
      "OpenVLA-OFT (PD&AC, Cont-L1)",
      "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only"
     ]
    },
    {
     "id": "libero-libero-spatial-in-distribution",
     "split": "LIBERO-Spatial",
     "condition": "in-distribution",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 10,
     "rows_with_n": 3,
     "best": {
      "policy": "OpenVLA-OFT (PD&AC, Cont-L1)",
      "value_percent": 97.6,
      "interval_low": 95.85,
      "interval_high": 98.61,
      "n_trials": 500,
      "row_id": "libero-libero-spatial-in-distribution-openvla-oft-pd-ac-cont-l1"
     },
     "headroom": 2.4,
     "leaders": [
      "OpenVLA-OFT (PD&AC, Cont-L1)",
      "OpenVLA-OFT (PD&AC, Cont-L1), third-person image only",
      "OpenVLA-OFT (PD&AC, Cont-L1), unfiltered dataset"
     ]
    },
    {
     "id": "libero-libero-spatial-site-bounded-trials",
     "split": "LIBERO-Spatial",
     "condition": "site bounded trials",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 1,
     "rows_with_n": 1,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 100.0,
      "interval_low": 22.36,
      "interval_high": 100.0,
      "n_trials": 1,
      "row_id": "site.libero-spatial-task0-seed7"
     },
     "headroom": 0.0,
     "leaders": [
      "OpenVLA-OFT"
     ]
    }
   ],
   "robustness": [],
   "sim_real": [],
   "evidence": {
    "rows": 43,
    "rows_with_n": 13,
    "evaluators": {
     "self": 34,
     "site": 1,
     "third-party": 8
    },
    "media": [
     "sim"
    ]
   }
  },
  {
   "benchmark": "LIBERO-Plus",
   "status": "saturated",
   "badges": [
    "saturated",
    "too-few-perturbed",
    "unvalidated",
    "n-missing"
   ],
   "catalogue": {
    "id": "sim--sim-manipulation-eval--libero",
    "name": "LIBERO",
    "first_public": "arXiv 2023-06-05 [R5]",
    "href": "/topics/benchmarks.html#sim--sim-manipulation-eval"
   },
   "boards": [
    {
     "id": "libero-plus-libero-all-four-suites-background-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Background perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 92.4,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-background-perturbation-openvla-oft"
     },
     "headroom": 7.6,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-camera-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Camera perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "π0-FAST",
      "value_percent": 66.4,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-camera-perturbation-pi0-fast"
     },
     "headroom": 33.6,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-in-distribution",
     "split": "LIBERO (all four suites)",
     "condition": "in-distribution",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 97.1,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-in-distribution-openvla-oft"
     },
     "headroom": 2.9,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-language-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Language perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 81.5,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-language-perturbation-openvla-oft"
     },
     "headroom": 18.5,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-layout-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Layout perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 77.1,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-layout-perturbation-openvla-oft"
     },
     "headroom": 22.9,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-light-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Light perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 85.8,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-light-perturbation-openvla-oft"
     },
     "headroom": 14.2,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-noise-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Noise perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "π0",
      "value_percent": 79.4,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-noise-perturbation-pi0"
     },
     "headroom": 20.6,
     "leaders": null
    },
    {
     "id": "libero-plus-libero-all-four-suites-robot-perturbation",
     "split": "LIBERO (all four suites)",
     "condition": "Robot perturbation",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 0,
     "best": {
      "policy": "OpenVLA-OFT",
      "value_percent": 37.2,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "libero-plus-libero-all-four-suites-robot-perturbation-openvla-oft"
     },
     "headroom": 62.8,
     "leaders": null
    }
   ],
   "robustness": [
    {
     "condition": "Robot perturbation",
     "board": "libero-plus-libero-all-four-suites-robot-perturbation",
     "mean_drop": 70.15,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "π0",
      "base": 94.2,
      "perturbed": 6.6,
      "drop": 87.6
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 4.1,
       "drop": 72.4
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 37.2,
       "drop": 59.9
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 6.6,
       "drop": 87.6
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 24.8,
       "drop": 60.7
      }
     ]
    },
    {
     "condition": "Camera perturbation",
     "board": "libero-plus-libero-all-four-suites-camera-perturbation",
     "mean_drop": 52.58,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "π0",
      "base": 94.2,
      "perturbed": 15.8,
      "drop": 78.4
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 1.1,
       "drop": 75.4
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 59.7,
       "drop": 37.4
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 15.8,
       "drop": 78.4
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 66.4,
       "drop": 19.1
      }
     ]
    },
    {
     "condition": "Language perturbation",
     "board": "libero-plus-libero-all-four-suites-language-perturbation",
     "mean_drop": 30.18,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "OpenVLA",
      "base": 76.5,
      "perturbed": 26.8,
      "drop": 49.7
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 26.8,
       "drop": 49.7
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 81.5,
       "drop": 15.6
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 61.0,
       "drop": 33.2
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 63.3,
       "drop": 22.2
      }
     ]
    },
    {
     "condition": "Light perturbation",
     "board": "libero-plus-libero-all-four-suites-light-perturbation",
     "mean_drop": 27.62,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "OpenVLA",
      "base": 76.5,
      "perturbed": 4.4,
      "drop": 72.1
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 4.4,
       "drop": 72.1
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 85.8,
       "drop": 11.3
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 79.6,
       "drop": 14.6
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 73.0,
       "drop": 12.5
      }
     ]
    },
    {
     "condition": "Layout perturbation",
     "board": "libero-plus-libero-all-four-suites-layout-perturbation",
     "mean_drop": 25.98,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "OpenVLA",
      "base": 76.5,
      "perturbed": 31.6,
      "drop": 44.9
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 31.6,
       "drop": 44.9
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 77.1,
       "drop": 20.0
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 70.4,
       "drop": 23.8
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 70.3,
       "drop": 15.2
      }
     ]
    },
    {
     "condition": "Noise perturbation",
     "board": "libero-plus-libero-all-four-suites-noise-perturbation",
     "mean_drop": 25.52,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "OpenVLA",
      "base": 76.5,
      "perturbed": 19.3,
      "drop": 57.2
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 19.3,
       "drop": 57.2
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 76.7,
       "drop": 20.4
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 79.4,
       "drop": 14.8
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 75.8,
       "drop": 9.7
      }
     ]
    },
    {
     "condition": "Background perturbation",
     "board": "libero-plus-libero-all-four-suites-background-perturbation",
     "mean_drop": 22.35,
     "verdict": "too-few-policies",
     "worst": {
      "policy": "OpenVLA",
      "base": 76.5,
      "perturbed": 25.3,
      "drop": 51.2
     },
     "policies": [
      {
       "policy": "OpenVLA",
       "base": 76.5,
       "perturbed": 25.3,
       "drop": 51.2
      },
      {
       "policy": "OpenVLA-OFT",
       "base": 97.1,
       "perturbed": 92.4,
       "drop": 4.7
      },
      {
       "policy": "π0",
       "base": 94.2,
       "perturbed": 78.5,
       "drop": 15.7
      },
      {
       "policy": "π0-FAST",
       "base": 85.5,
       "perturbed": 67.7,
       "drop": 17.8
      }
     ]
    }
   ],
   "sim_real": [],
   "evidence": {
    "rows": 32,
    "rows_with_n": 0,
    "evaluators": {
     "third-party": 32
    },
    "media": [
     "sim"
    ]
   }
  },
  {
   "benchmark": "BEHAVIOR Challenge 2025",
   "status": "open",
   "badges": [
    "unvalidated",
    "open",
    "leaders-tied"
   ],
   "catalogue": {
    "id": "activity--everyday-activity-sim--behavior-1k",
    "name": "BEHAVIOR-1K",
    "first_public": "arXiv 2024-03-14 [R16]",
    "href": "/topics/benchmarks.html#activity--everyday-activity-sim"
   },
   "boards": [
    {
     "id": "behavior-challenge-2025-held-out-test-50-tasks-privileged-track",
     "split": "held-out test, 50 tasks",
     "condition": "privileged track",
     "metric": "full-task success rate",
     "medium": [
      "sim"
     ],
     "rows": 1,
     "rows_with_n": 1,
     "best": {
      "policy": "Embodied Intelligence",
      "value_percent": 5.2,
      "interval_low": 3.58,
      "interval_high": 7.51,
      "n_trials": 500,
      "row_id": "behavior-challenge-2025-held-out-test-50-tasks-privileged-track-embodied-intelligence"
     },
     "headroom": 94.8,
     "leaders": [
      "Embodied Intelligence"
     ]
    },
    {
     "id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track",
     "split": "held-out test, 50 tasks",
     "condition": "standard track",
     "metric": "full-task success rate",
     "medium": [
      "sim"
     ],
     "rows": 4,
     "rows_with_n": 4,
     "best": {
      "policy": "Robot Learning Collective",
      "value_percent": 12.4,
      "interval_low": 9.8,
      "interval_high": 15.58,
      "n_trials": 500,
      "row_id": "behavior-challenge-2025-held-out-test-50-tasks-standard-track-robot-learning-collective"
     },
     "headroom": 87.6,
     "leaders": [
      "Comet (NVIDIA Research)",
      "Robot Learning Collective",
      "SimpleAI Robot",
      "The North Star (Huawei CRI EAI)"
     ]
    }
   ],
   "robustness": [],
   "sim_real": [],
   "evidence": {
    "rows": 5,
    "rows_with_n": 5,
    "evaluators": {
     "challenge": 5
    },
    "media": [
     "sim"
    ]
   }
  },
  {
   "benchmark": "RoboChallenge Table30",
   "status": "open",
   "badges": [
    "real-robot",
    "open"
   ],
   "catalogue": null,
   "boards": [
    {
     "id": "robochallenge-table30-30-tasks-average-generalist-multi-task",
     "split": "30 tasks, average",
     "condition": "generalist multi-task",
     "metric": "success rate",
     "medium": [
      "real"
     ],
     "rows": 2,
     "rows_with_n": 2,
     "best": {
      "policy": "π0.5/multi",
      "value_percent": 17.7,
      "interval_low": 13.77,
      "interval_high": 22.39,
      "n_trials": 300,
      "row_id": "robochallenge-table30-30-tasks-average-generalist-multi-task-pi0-5-multi"
     },
     "headroom": 82.3,
     "leaders": [
      "π0.5/multi"
     ]
    },
    {
     "id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune",
     "split": "30 tasks, average",
     "condition": "task-specific fine-tune",
     "metric": "success rate",
     "medium": [
      "real"
     ],
     "rows": 3,
     "rows_with_n": 3,
     "best": {
      "policy": "π0.5",
      "value_percent": 43.7,
      "interval_low": 38.17,
      "interval_high": 49.33,
      "n_trials": 300,
      "row_id": "robochallenge-table30-30-tasks-average-task-specific-fine-tune-pi0-5"
     },
     "headroom": 56.3,
     "leaders": [
      "π0.5"
     ]
    }
   ],
   "robustness": [],
   "sim_real": [],
   "evidence": {
    "rows": 5,
    "rows_with_n": 5,
    "evaluators": {
     "challenge": 5
    },
    "media": [
     "real"
    ]
   }
  },
  {
   "benchmark": "SimplerEnv WidowX",
   "status": "open",
   "badges": [
    "too-few-policies",
    "real-robot",
    "open",
    "leaders-tied"
   ],
   "catalogue": {
    "id": "simpler--real-to-sim-eval--widowx",
    "name": "WidowX",
    "first_public": "arXiv 2024-05-09 [R13]",
    "href": "/topics/benchmarks.html#simpler--real-to-sim-eval"
   },
   "boards": [
    {
     "id": "simplerenv-widowx-put-spoon-on-towel-real-robot",
     "split": "Put Spoon on Towel",
     "condition": "real robot",
     "metric": "success rate",
     "medium": [
      "real"
     ],
     "rows": 3,
     "rows_with_n": 3,
     "best": {
      "policy": "Octo-Small",
      "value_percent": 41.7,
      "interval_low": 24.4,
      "interval_high": 61.33,
      "n_trials": 24,
      "row_id": "simplerenv-widowx-put-spoon-on-towel-real-robot-octo-small"
     },
     "headroom": 58.3,
     "leaders": [
      "Octo-Base",
      "Octo-Small"
     ]
    },
    {
     "id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching",
     "split": "Put Spoon on Towel",
     "condition": "SIMPLER visual matching",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 3,
     "rows_with_n": 0,
     "best": {
      "policy": "Octo-Small",
      "value_percent": 47.2,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching-octo-small"
     },
     "headroom": 52.8,
     "leaders": null
    },
    {
     "id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot",
     "split": "Stack Green Block on Yellow Block",
     "condition": "real robot",
     "metric": "success rate",
     "medium": [
      "real"
     ],
     "rows": 3,
     "rows_with_n": 3,
     "best": {
      "policy": "Octo-Small",
      "value_percent": 12.5,
      "interval_low": 4.54,
      "interval_high": 31.22,
      "n_trials": 24,
      "row_id": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot-octo-small"
     },
     "headroom": 87.5,
     "leaders": [
      "Octo-Base",
      "Octo-Small",
      "RT-1-X"
     ]
    },
    {
     "id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching",
     "split": "Stack Green Block on Yellow Block",
     "condition": "SIMPLER visual matching",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 3,
     "rows_with_n": 0,
     "best": {
      "policy": "Octo-Small",
      "value_percent": 4.2,
      "interval_low": null,
      "interval_high": null,
      "n_trials": null,
      "row_id": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching-octo-small"
     },
     "headroom": 95.8,
     "leaders": null
    }
   ],
   "robustness": [],
   "sim_real": [
    {
     "split": "Put Spoon on Towel",
     "sim_board": "simplerenv-widowx-put-spoon-on-towel-simpler-visual-matching",
     "real_board": "simplerenv-widowx-put-spoon-on-towel-real-robot",
     "policies": [
      {
       "policy": "Octo-Base",
       "sim": 12.5,
       "real": 33.3
      },
      {
       "policy": "Octo-Small",
       "sim": 47.2,
       "real": 41.7
      },
      {
       "policy": "RT-1-X",
       "sim": 0.0,
       "real": 0.0
      }
     ],
     "spearman": 1.0,
     "mmrv": 0.0,
     "mean_abs_gap": 8.77,
     "real_separates": true,
     "verdict": "too-few-policies"
    },
    {
     "split": "Stack Green Block on Yellow Block",
     "sim_board": "simplerenv-widowx-stack-green-block-on-yellow-block-simpler-visual-matching",
     "real_board": "simplerenv-widowx-stack-green-block-on-yellow-block-real-robot",
     "policies": [
      {
       "policy": "Octo-Base",
       "sim": 0.0,
       "real": 0.0
      },
      {
       "policy": "Octo-Small",
       "sim": 4.2,
       "real": 12.5
      },
      {
       "policy": "RT-1-X",
       "sim": 0.0,
       "real": 0.0
      }
     ],
     "spearman": 1.0,
     "mmrv": 0.0,
     "mean_abs_gap": 2.77,
     "real_separates": false,
     "verdict": "too-few-policies"
    }
   ],
   "evidence": {
    "rows": 12,
    "rows_with_n": 6,
    "evaluators": {
     "third-party": 12
    },
    "media": [
     "real",
     "sim"
    ]
   }
  },
  {
   "benchmark": "ManiSkill3",
   "status": "unvalidated",
   "badges": [
    "unvalidated"
   ],
   "catalogue": null,
   "boards": [
    {
     "id": "maniskill3-pickcube-v1-reference-seeds",
     "split": "PickCube-v1",
     "condition": "reference seeds",
     "metric": "success rate",
     "medium": [
      "sim"
     ],
     "rows": 1,
     "rows_with_n": 1,
     "best": {
      "policy": "ManiSkill PPO reference checkpoint",
      "value_percent": 100.0,
      "interval_low": 76.16,
      "interval_high": 100.0,
      "n_trials": 10,
      "row_id": "site.maniskill3.pickcube.ppo"
     },
     "headroom": 0.0,
     "leaders": [
      "ManiSkill PPO reference checkpoint"
     ]
    }
   ],
   "robustness": [],
   "sim_real": [],
   "evidence": {
    "rows": 1,
    "rows_with_n": 1,
    "evaluators": {
     "site": 1
    },
    "media": [
     "sim"
    ]
   }
  }
 ]
}
