{
 "$schema": "https://botlandscape.com/benchmark-history.schema.json",
 "schema_version": 1,
 "method": {
  "version": "benchmark-history/1",
  "frontier": "Progress rows sorted by date; a point joins the frontier when it beats every earlier point, and carries the id of its progress row. Values are divided by the suite's scale maximum. Headroom is 100 minus the best.",
  "robustness": "retained = perturbed / clean for one policy under one perturbation in one source; median and worst over the suite's rows.",
  "fidelity": "Agreement statistics copied as published, with the real-robot evaluation they were compared against. MMRV is a ranking-violation measure where lower is better. A statement is a published claim with no number. Proxies are real-to-sim evaluators outside the catalogue, listed for comparison.",
  "dates": "A point is plotted at method_date, the first arXiv version of the paper that introduced the method, when it is recorded; otherwise at date, the first public version of the source that prints the number. A point is never plotted before the suite's own first public date in the catalogue. The score itself always comes from the cited source.",
  "scope": "Scores are copied as their sources print them. Progress points mix self-reported and third-party numbers and are marked. Badges live on catalogue/benchmark-health.json; this file sets none."
 },
 "suites": [
  {
   "id": "sim--sim-manipulation-eval--libero",
   "name": "LIBERO",
   "metric": "average success rate, 4 suites (Spatial/Object/Goal/Long)",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#sim--sim-manipulation-eval--libero",
   "released": "2023-06-05",
   "note": "",
   "ceiling": null,
   "best_percent": 98.8,
   "headroom_percent": 1.2,
   "first_date": "2023-06-05",
   "latest_date": "2026-08-12",
   "median_retained": 0.6342,
   "worst_retained": 0.0144,
   "frontier": [
    {
     "date": "2023-06-05",
     "percent": 72.4,
     "policy": "Diffusion Policy (scratch)",
     "id": "libero-pro-diffusion-policy-scratch-2025-02-27"
    },
    {
     "date": "2024-05-20",
     "percent": 75.1,
     "policy": "Octo (fine-tuned)",
     "id": "libero-pro-octo-fine-tuned-2025-02-27"
    },
    {
     "date": "2024-06-13",
     "percent": 76.5,
     "policy": "OpenVLA (fine-tuned)",
     "id": "libero-pro-openvla-fine-tuned-2025-02-27"
    },
    {
     "date": "2024-10-31",
     "percent": 94.2,
     "policy": "pi0 (fine-tuned)",
     "id": "libero-pro-pi0-fine-tuned-2025-02-27"
    },
    {
     "date": "2025-02-27",
     "percent": 97.1,
     "policy": "OpenVLA-OFT",
     "id": "libero-pro-openvla-oft-2025-02-27"
    },
    {
     "date": "2025-10-11",
     "percent": 98.1,
     "policy": "X-VLA-0.9B",
     "id": "libero-pro-x-vla-0-9b-2025-10-11"
    },
    {
     "date": "2026-01-16",
     "percent": 98.5,
     "policy": "ACoT-VLA",
     "id": "libero-pro-acot-vla-2026-08-12"
    },
    {
     "date": "2026-08-12",
     "percent": 98.8,
     "policy": "StellaVLA",
     "id": "libero-pro-stellavla-2026-08-12"
    }
   ],
   "progress": [
    {
     "policy": "Diffusion Policy (scratch)",
     "date": "2025-02-27",
     "value": 72.4,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2502.19645",
     "locator": "Table I (LIBERO task performance results)",
     "quote": "Diffusion Policy (scratch) [5] 78.3 92.5 68.3 50.5 72.4",
     "percent": 72.4,
     "method_date": "2023-03-07",
     "method_url": "https://arxiv.org/abs/2303.04137",
     "plot_date": "2023-06-05",
     "note": "",
     "id": "libero-pro-diffusion-policy-scratch-2025-02-27"
    },
    {
     "policy": "Octo (fine-tuned)",
     "date": "2025-02-27",
     "value": 75.1,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2502.19645",
     "locator": "Table I (LIBERO task performance results)",
     "quote": "Octo (fine-tuned) [49] 78.9 85.7 84.6 51.1 75.1",
     "percent": 75.1,
     "method_date": "2024-05-20",
     "method_url": "https://arxiv.org/abs/2405.12213",
     "plot_date": "2024-05-20",
     "note": "",
     "id": "libero-pro-octo-fine-tuned-2025-02-27"
    },
    {
     "policy": "OpenVLA (fine-tuned)",
     "date": "2025-02-27",
     "value": 76.5,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2502.19645",
     "locator": "Table I (LIBERO task performance results)",
     "quote": "OpenVLA (fine-tuned) [23] ... 88.4 79.2 53.7 76.5",
     "percent": 76.5,
     "method_date": "2024-06-13",
     "method_url": "https://arxiv.org/abs/2406.09246",
     "plot_date": "2024-06-13",
     "note": "OpenVLA's own paper (2406.09246, v1 2024-06-13) reports this in App. E, which WebFetch could not reach; value read from OFT Table I (OFT shares first author with OpenVLA).",
     "id": "libero-pro-openvla-fine-tuned-2025-02-27"
    },
    {
     "policy": "pi0 (fine-tuned)",
     "date": "2025-02-27",
     "value": 94.2,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2502.19645",
     "locator": "Table I (LIBERO task performance results)",
     "quote": "π0 (fine-tuned) [3] 96.8 98.8 95.8 85.2 94.2",
     "percent": 94.2,
     "method_date": "2024-10-31",
     "method_url": "https://arxiv.org/abs/2410.24164",
     "plot_date": "2024-10-31",
     "note": "",
     "id": "libero-pro-pi0-fine-tuned-2025-02-27"
    },
    {
     "policy": "OpenVLA-OFT",
     "date": "2025-02-27",
     "value": 97.1,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2502.19645",
     "locator": "Table I (LIBERO task performance results)",
     "quote": "OpenVLA-OFT (ours) 97.6 98.4 97.9 94.5 97.1",
     "percent": 97.1,
     "method_date": "2025-02-27",
     "method_url": "https://arxiv.org/abs/2502.19645",
     "plot_date": "2025-02-27",
     "note": "",
     "id": "libero-pro-openvla-oft-2025-02-27"
    },
    {
     "policy": "X-VLA-0.9B",
     "date": "2025-10-11",
     "value": 98.1,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2510.10274",
     "locator": "Table 2 (LIBERO Avg column)",
     "quote": "X-VLA (Ours) 0.9B 80.4 75.7 95.8 98.2 98.6 97.8 97.6 98.1 ...",
     "percent": 98.1,
     "method_date": "2025-10-11",
     "method_url": "https://arxiv.org/abs/2510.10274",
     "plot_date": "2025-10-11",
     "note": "",
     "id": "libero-pro-x-vla-0-9b-2025-10-11"
    },
    {
     "policy": "ACoT-VLA",
     "date": "2026-08-12",
     "value": 98.5,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2608.11671",
     "locator": "Table 1",
     "quote": "ACoT-VLA 99.4 99.6 98.8 96.0 98.5",
     "percent": 98.5,
     "method_date": "2026-01-16",
     "method_url": "https://arxiv.org/abs/2601.11404",
     "plot_date": "2026-01-16",
     "note": "Number as reported by the StellaVLA authors, who say baselines are copied from original papers; ACoT-VLA paper itself not inspected.",
     "id": "libero-pro-acot-vla-2026-08-12"
    },
    {
     "policy": "StellaVLA",
     "date": "2026-08-12",
     "value": 98.8,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2608.11671",
     "locator": "Table 1",
     "quote": "StellaVLA (Ours) 99.6 99.0 99.6 96.8 98.8",
     "percent": 98.8,
     "method_date": "2026-08-12",
     "method_url": "https://arxiv.org/abs/2608.11671",
     "plot_date": "2026-08-12",
     "note": "",
     "id": "libero-pro-stellavla-2026-08-12"
    }
   ],
   "robustness_rows": [
    {
     "perturbation_suite": "LIBERO-PRO",
     "perturbation": "object position (libero-goal)",
     "policy": "pi0.5",
     "clean_value": 97.0,
     "perturbed_value": 38.0,
     "source_url": "https://arxiv.org/abs/2510.03827",
     "locator": "Sec 5.2 (perturbed); clean value from GitHub README results badges",
     "quote": "Pi0.5 achieves a 0.38 success rate in the libero-goal task under position changes",
     "clean_percent": 97.0,
     "perturbed_percent": 38.0,
     "drop_points": 59.0,
     "retained": 0.3918,
     "note": "Clean 0.97 decoded from README shield badges (not verbatim HTML text).",
     "id": "libero-rob-object-position-libero-goal-pi0-5"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "background",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 25.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 25.3,
     "drop_points": 51.2,
     "retained": 0.3307,
     "note": "",
     "id": "libero-rob-background-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "background",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 67.7,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 67.7,
     "drop_points": 17.8,
     "retained": 0.7918,
     "note": "",
     "id": "libero-rob-background-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "background",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 78.5,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 78.5,
     "drop_points": 15.7,
     "retained": 0.8333,
     "note": "",
     "id": "libero-rob-background-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "background",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 80.0,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 80.0,
     "drop_points": 15.2,
     "retained": 0.8403,
     "note": "",
     "id": "libero-rob-background-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "background",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 92.4,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 92.4,
     "drop_points": 4.7,
     "retained": 0.9516,
     "note": "",
     "id": "libero-rob-background-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "camera viewpoint",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 1.1,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 1.1,
     "drop_points": 75.4,
     "retained": 0.0144,
     "note": "",
     "id": "libero-rob-camera-viewpoint-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "camera viewpoint",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 4.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 4.3,
     "drop_points": 90.9,
     "retained": 0.0452,
     "note": "",
     "id": "libero-rob-camera-viewpoint-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "camera viewpoint",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 15.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 15.8,
     "drop_points": 78.4,
     "retained": 0.1677,
     "note": "",
     "id": "libero-rob-camera-viewpoint-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "camera viewpoint",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 59.7,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 59.7,
     "drop_points": 37.4,
     "retained": 0.6148,
     "note": "",
     "id": "libero-rob-camera-viewpoint-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "camera viewpoint",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 66.4,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 66.4,
     "drop_points": 19.1,
     "retained": 0.7766,
     "note": "",
     "id": "libero-rob-camera-viewpoint-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "language",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 26.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 26.8,
     "drop_points": 49.7,
     "retained": 0.3503,
     "note": "",
     "id": "libero-rob-language-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "language",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 61.0,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 61.0,
     "drop_points": 33.2,
     "retained": 0.6476,
     "note": "",
     "id": "libero-rob-language-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "language",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 63.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 63.3,
     "drop_points": 22.2,
     "retained": 0.7404,
     "note": "",
     "id": "libero-rob-language-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "language",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 71.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 71.8,
     "drop_points": 23.4,
     "retained": 0.7542,
     "note": "",
     "id": "libero-rob-language-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "language",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 81.5,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 81.5,
     "drop_points": 15.6,
     "retained": 0.8393,
     "note": "",
     "id": "libero-rob-language-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "layout",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 34.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 34.3,
     "drop_points": 60.9,
     "retained": 0.3603,
     "note": "",
     "id": "libero-rob-layout-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "layout",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 31.6,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 31.6,
     "drop_points": 44.9,
     "retained": 0.4131,
     "note": "",
     "id": "libero-rob-layout-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "layout",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 70.4,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 70.4,
     "drop_points": 23.8,
     "retained": 0.7473,
     "note": "",
     "id": "libero-rob-layout-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "layout",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 77.1,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 77.1,
     "drop_points": 20.0,
     "retained": 0.794,
     "note": "",
     "id": "libero-rob-layout-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "layout",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 70.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 70.3,
     "drop_points": 15.2,
     "retained": 0.8222,
     "note": "",
     "id": "libero-rob-layout-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "light",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 4.4,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 4.4,
     "drop_points": 72.1,
     "retained": 0.0575,
     "note": "",
     "id": "libero-rob-light-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "light",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 59.1,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 59.1,
     "drop_points": 36.1,
     "retained": 0.6208,
     "note": "",
     "id": "libero-rob-light-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "light",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 79.6,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 79.6,
     "drop_points": 14.6,
     "retained": 0.845,
     "note": "",
     "id": "libero-rob-light-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "light",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 73.0,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 73.0,
     "drop_points": 12.5,
     "retained": 0.8538,
     "note": "",
     "id": "libero-rob-light-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "light",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 85.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 85.8,
     "drop_points": 11.3,
     "retained": 0.8836,
     "note": "",
     "id": "libero-rob-light-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "robot initial state",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 4.1,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 4.1,
     "drop_points": 72.4,
     "retained": 0.0536,
     "note": "",
     "id": "libero-rob-robot-initial-state-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "robot initial state",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 6.6,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 6.6,
     "drop_points": 87.6,
     "retained": 0.0701,
     "note": "",
     "id": "libero-rob-robot-initial-state-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "robot initial state",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 24.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 24.8,
     "drop_points": 60.7,
     "retained": 0.2901,
     "note": "",
     "id": "libero-rob-robot-initial-state-pi0-fast"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "robot initial state",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 37.2,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 37.2,
     "drop_points": 59.9,
     "retained": 0.3831,
     "note": "",
     "id": "libero-rob-robot-initial-state-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "robot initial state",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 50.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 50.3,
     "drop_points": 44.9,
     "retained": 0.5284,
     "note": "",
     "id": "libero-rob-robot-initial-state-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "sensor noise",
     "policy": "OpenVLA",
     "clean_value": 76.5,
     "perturbed_value": 19.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA 76.5 1.1 4.1 26.8 4.4 25.3 19.3 31.6",
     "clean_percent": 76.5,
     "perturbed_percent": 19.3,
     "drop_points": 57.2,
     "retained": 0.2523,
     "note": "",
     "id": "libero-rob-sensor-noise-openvla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "sensor noise",
     "policy": "UniVLA",
     "clean_value": 95.2,
     "perturbed_value": 25.3,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "UniVLA 95.2 4.3 50.3 71.8 59.1 80.0 25.3 34.3",
     "clean_percent": 95.2,
     "perturbed_percent": 25.3,
     "drop_points": 69.9,
     "retained": 0.2658,
     "note": "",
     "id": "libero-rob-sensor-noise-univla"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "sensor noise",
     "policy": "OpenVLA-OFT",
     "clean_value": 97.1,
     "perturbed_value": 76.7,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "OpenVLA-OFT 97.1 59.7 37.2 81.5 85.8 92.4 76.7 77.1",
     "clean_percent": 97.1,
     "perturbed_percent": 76.7,
     "drop_points": 20.4,
     "retained": 0.7899,
     "note": "",
     "id": "libero-rob-sensor-noise-openvla-oft"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "sensor noise",
     "policy": "pi0",
     "clean_value": 94.2,
     "perturbed_value": 79.4,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0 94.2 15.8 6.6 61.0 79.6 78.5 79.4 70.4",
     "clean_percent": 94.2,
     "perturbed_percent": 79.4,
     "drop_points": 14.8,
     "retained": 0.8429,
     "note": "",
     "id": "libero-rob-sensor-noise-pi0"
    },
    {
     "perturbation_suite": "LIBERO-Plus",
     "perturbation": "sensor noise",
     "policy": "pi0-FAST",
     "clean_value": 85.5,
     "perturbed_value": 75.8,
     "source_url": "https://arxiv.org/abs/2510.13626",
     "locator": "Table 1 (Model performance under different perturbations.); columns Original, Camera, Robot, Language, Light, Background, Noise, Layout",
     "quote": "π0-fast 85.5 66.4 24.8 63.3 73.0 67.7 75.8 70.3",
     "clean_percent": 85.5,
     "perturbed_percent": 75.8,
     "drop_points": 9.7,
     "retained": 0.8865,
     "note": "",
     "id": "libero-rob-sensor-noise-pi0-fast"
    }
   ],
   "fidelity_rows": [
    {
     "statistic": "statement",
     "real_reference": "real DROID evaluations",
     "source_url": "https://arxiv.org/abs/2512.16881",
     "locator": "Sec 5.2",
     "quote": "Libero-score too is poorly correlated with real-world performance",
     "value": null,
     "n_policies": 4,
     "note": "Qualitative; policies fine-tuned on Libero-90 scored nearly all 90–95% in Libero (per page text)."
    }
   ]
  },
  {
   "id": "sim--sim-manipulation-eval--calvin",
   "name": "CALVIN",
   "metric": "average successful sequence length, ABC→D (5-task chains)",
   "unit": "avg len /5",
   "scale_max": 5,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#sim--sim-manipulation-eval--calvin",
   "released": "2021-12-06",
   "note": "",
   "ceiling": null,
   "best_percent": 88.6,
   "headroom_percent": 11.4,
   "first_date": "2021-12-06",
   "latest_date": "2025-10-11",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2021-12-06",
     "percent": 6.2,
     "policy": "MCIL",
     "id": "calvin-pro-mcil-2023-12-20"
    },
    {
     "date": "2022-04-13",
     "percent": 13.4,
     "policy": "HULC",
     "id": "calvin-pro-hulc-2023-12-20"
    },
    {
     "date": "2023-11-02",
     "percent": 49.4,
     "policy": "RoboFlamingo",
     "id": "calvin-pro-roboflamingo-2024-12-19"
    },
    {
     "date": "2023-12-20",
     "percent": 61.2,
     "policy": "GR-1",
     "id": "calvin-pro-gr-1-2023-12-20"
    },
    {
     "date": "2024-02-16",
     "percent": 65.4,
     "policy": "3D Diffuser Actor",
     "id": "calvin-pro-3d-diffuser-actor-2024-12-19"
    },
    {
     "date": "2024-12-19",
     "percent": 85.6,
     "policy": "Seer-Large",
     "id": "calvin-pro-seer-large-2024-12-19"
    },
    {
     "date": "2025-10-11",
     "percent": 88.6,
     "policy": "X-VLA-0.9B",
     "id": "calvin-pro-x-vla-0-9b-2025-10-11"
    }
   ],
   "progress": [
    {
     "policy": "MCIL",
     "date": "2023-12-20",
     "value": 0.31,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2312.13139",
     "locator": "Table 1 (CALVIN Benchmark Results), ABC→D",
     "quote": "MCIL ABC→D 0.304 0.013 0.002 0.000 0.000 0.31",
     "percent": 6.2,
     "method_date": "2020-05-15",
     "method_url": "https://arxiv.org/abs/2005.07648",
     "plot_date": "2021-12-06",
     "note": "",
     "id": "calvin-pro-mcil-2023-12-20"
    },
    {
     "policy": "HULC",
     "date": "2023-12-20",
     "value": 0.67,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2312.13139",
     "locator": "Table 1, ABC→D",
     "quote": "HULC ABC→D 0.418 0.165 0.057 0.019 0.011 0.67",
     "percent": 13.4,
     "method_date": "2022-04-13",
     "method_url": "https://arxiv.org/abs/2204.06252",
     "plot_date": "2022-04-13",
     "note": "",
     "id": "calvin-pro-hulc-2023-12-20"
    },
    {
     "policy": "RoboFlamingo",
     "date": "2024-12-19",
     "value": 2.47,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2412.15109",
     "locator": "Table 2 (CALVIN ABC-D results)",
     "quote": "Roboflamingo 82.4 61.9 46.6 33.1 23.5 2.47",
     "percent": 49.4,
     "method_date": "2023-11-02",
     "method_url": "https://arxiv.org/abs/2311.01378",
     "plot_date": "2023-11-02",
     "note": "",
     "id": "calvin-pro-roboflamingo-2024-12-19"
    },
    {
     "policy": "GR-1",
     "date": "2023-12-20",
     "value": 3.06,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2312.13139",
     "locator": "Table 1, ABC→D",
     "quote": "GR-1 (Ours) ABC→D 0.854 0.712 0.596 0.497 0.401 3.06",
     "percent": 61.2,
     "method_date": "2023-12-20",
     "method_url": "https://arxiv.org/abs/2312.13139",
     "plot_date": "2023-12-20",
     "note": "",
     "id": "calvin-pro-gr-1-2023-12-20"
    },
    {
     "policy": "3D Diffuser Actor",
     "date": "2024-12-19",
     "value": 3.27,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2412.15109",
     "locator": "Table 2",
     "quote": "3D Diffusor Actor 92.2 78.7 63.9 51.2 41.2 3.27",
     "percent": 65.4,
     "method_date": "2024-02-16",
     "method_url": "https://arxiv.org/abs/2402.10885",
     "plot_date": "2024-02-16",
     "note": "",
     "id": "calvin-pro-3d-diffuser-actor-2024-12-19"
    },
    {
     "policy": "Seer-Large",
     "date": "2024-12-19",
     "value": 4.28,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2412.15109",
     "locator": "Table 2",
     "quote": "Seer-Large 96.3 91.6 86.1 80.3 74.0 4.28",
     "percent": 85.6,
     "method_date": "2024-12-19",
     "method_url": "https://arxiv.org/abs/2412.15109",
     "plot_date": "2024-12-19",
     "note": "",
     "id": "calvin-pro-seer-large-2024-12-19"
    },
    {
     "policy": "X-VLA-0.9B",
     "date": "2025-10-11",
     "value": 4.43,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2510.10274",
     "locator": "Table 2 (CALVIN ABC→D column)",
     "quote": "X-VLA (Ours) 0.9B ... 98.1 4.43 70.0 39.0 51.1 87.3",
     "percent": 88.6,
     "method_date": "2025-10-11",
     "method_url": "https://arxiv.org/abs/2510.10274",
     "plot_date": "2025-10-11",
     "note": "Same table's 'Maximum of Existing SOTA' row lists 4.53 for CALVIN without naming the method; not recorded as a row.",
     "id": "calvin-pro-x-vla-0-9b-2025-10-11"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  },
  {
   "id": "sim--sim-manipulation-eval--rlbench",
   "name": "RLBench",
   "metric": "average success rate, 18 RLBench tasks (multi-task, PerAct split)",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#sim--sim-manipulation-eval--rlbench",
   "released": "2019-09-26",
   "note": "",
   "ceiling": null,
   "best_percent": 86.8,
   "headroom_percent": 13.2,
   "first_date": "2021-06-23",
   "latest_date": "2025-01-30",
   "median_retained": 0.1855,
   "worst_retained": 0.1855,
   "frontier": [
    {
     "date": "2021-06-23",
     "percent": 20.1,
     "policy": "C2F-ARM-BC",
     "id": "rlbench-pro-c2f-arm-bc-2025-01-30"
    },
    {
     "date": "2022-09-12",
     "percent": 49.4,
     "policy": "PerAct",
     "id": "rlbench-pro-peract-2025-01-30"
    },
    {
     "date": "2023-06-26",
     "percent": 62.9,
     "policy": "RVT",
     "id": "rlbench-pro-rvt-2025-01-30"
    },
    {
     "date": "2024-02-16",
     "percent": 81.3,
     "policy": "3D Diffuser Actor",
     "id": "rlbench-pro-3d-diffuser-actor-2025-01-30"
    },
    {
     "date": "2024-06-12",
     "percent": 81.4,
     "policy": "RVT-2",
     "id": "rlbench-pro-rvt-2-2025-01-30"
    },
    {
     "date": "2024-10-04",
     "percent": 84.9,
     "policy": "ARP+",
     "id": "rlbench-pro-arp-2025-01-30"
    },
    {
     "date": "2025-01-30",
     "percent": 86.8,
     "policy": "SAM2Act",
     "id": "rlbench-pro-sam2act-2025-01-30"
    }
   ],
   "progress": [
    {
     "policy": "C2F-ARM-BC",
     "date": "2025-01-30",
     "value": 20.1,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "C2F-ARM-BC 20.1 (avg success) 11.5 (avg rank)",
     "percent": 20.1,
     "method_date": "2021-06-23",
     "method_url": "https://arxiv.org/abs/2106.12534",
     "plot_date": "2021-06-23",
     "note": "",
     "id": "rlbench-pro-c2f-arm-bc-2025-01-30"
    },
    {
     "policy": "PerAct",
     "date": "2025-01-30",
     "value": 49.4,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "PerAct 49.4 ± 4.3 ... 8.9",
     "percent": 49.4,
     "method_date": "2022-09-12",
     "method_url": "https://arxiv.org/abs/2209.05451",
     "plot_date": "2022-09-12",
     "note": "",
     "id": "rlbench-pro-peract-2025-01-30"
    },
    {
     "policy": "RVT",
     "date": "2025-01-30",
     "value": 62.9,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "RVT 62.9 ± 3.7 ... 6.9",
     "percent": 62.9,
     "method_date": "2023-06-26",
     "method_url": "https://arxiv.org/abs/2306.14896",
     "plot_date": "2023-06-26",
     "note": "",
     "id": "rlbench-pro-rvt-2025-01-30"
    },
    {
     "policy": "3D Diffuser Actor",
     "date": "2025-01-30",
     "value": 81.3,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "3D Diffuser Actor 81.3 ... 3.9",
     "percent": 81.3,
     "method_date": "2024-02-16",
     "method_url": "https://arxiv.org/abs/2402.10885",
     "plot_date": "2024-02-16",
     "note": "",
     "id": "rlbench-pro-3d-diffuser-actor-2025-01-30"
    },
    {
     "policy": "RVT-2",
     "date": "2025-01-30",
     "value": 81.4,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "RVT-2 81.4 ± 3.1 ... 3.7",
     "percent": 81.4,
     "method_date": "2024-06-12",
     "method_url": "https://arxiv.org/abs/2406.08545",
     "plot_date": "2024-06-12",
     "note": "",
     "id": "rlbench-pro-rvt-2-2025-01-30"
    },
    {
     "policy": "ARP+",
     "date": "2025-01-30",
     "value": 84.9,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "ARP+ 84.9 ... 3.2",
     "percent": 84.9,
     "method_date": "2024-10-04",
     "method_url": "https://arxiv.org/abs/2410.03132",
     "plot_date": "2024-10-04",
     "note": "",
     "id": "rlbench-pro-arp-2025-01-30"
    },
    {
     "policy": "SAM2Act",
     "date": "2025-01-30",
     "value": 86.8,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2501.18564",
     "locator": "Appendix C, Table 7",
     "quote": "SAM2Act 86.8 ± 0.5 ... 3.1",
     "percent": 86.8,
     "method_date": "2025-01-30",
     "method_url": "https://arxiv.org/abs/2501.18564",
     "plot_date": "2025-01-30",
     "note": "",
     "id": "rlbench-pro-sam2act-2025-01-30"
    }
   ],
   "robustness_rows": [
    {
     "perturbation_suite": "COLOSSEUM",
     "perturbation": "all perturbations (zero-shot)",
     "policy": "PerAct",
     "clean_value": 34.5,
     "perturbed_value": 6.4,
     "source_url": "https://arxiv.org/abs/2402.08191",
     "locator": "Sec V-B",
     "quote": "achieves task-averaged success rate of 6.4% (28.1% lower than No Perturbations task-averaged success rate)",
     "clean_percent": 34.5,
     "perturbed_percent": 6.4,
     "drop_points": 28.1,
     "retained": 0.1855,
     "note": "clean_value 34.5 derived as 6.4 + 28.1, not printed. Clean value derived, not printed.",
     "id": "rlbench-rob-all-perturbations-zero-shot-peract"
    }
   ],
   "fidelity_rows": [
    {
     "statistic": "r_squared",
     "real_reference": "real-world experiments with similar perturbations",
     "source_url": "https://arxiv.org/abs/2402.08191",
     "locator": "Abstract",
     "quote": "we show that our results in simulation are correlated (R̄²=0.614) to similar perturbations in real-world experiments.",
     "value": 0.614,
     "n_policies": null,
     "note": "COLOSSEUM, the perturbed RLBench variant. Mean R² across perturbation factors."
    }
   ]
  },
  {
   "id": "simpler--real-to-sim-eval--google-robot",
   "name": "Google Robot",
   "metric": "average success, 4 Google Robot tasks (Pick Coke Can, Move Near, Open/Close Drawer, Open Top Drawer & Place Apple), Visual Matching",
   "unit": "percent",
   "scale_max": 100,
   "medium": "paired",
   "catalogue_href": "/topics/benchmarks.html#simpler--real-to-sim-eval--google-robot",
   "released": "2024-05-09",
   "note": "",
   "ceiling": null,
   "best_percent": 74.8,
   "headroom_percent": 25.2,
   "first_date": "2024-05-09",
   "latest_date": "2024-11-29",
   "median_retained": 0.8267,
   "worst_retained": 0.1091,
   "frontier": [
    {
     "date": "2024-05-09",
     "percent": 52.4,
     "policy": "RT-1",
     "id": "google-robot-pro-rt-1-2024-11-29"
    },
    {
     "date": "2024-11-29",
     "percent": 74.8,
     "policy": "CogACT",
     "id": "google-robot-pro-cogact-2024-11-29"
    }
   ],
   "progress": [
    {
     "policy": "RT-1",
     "date": "2024-11-29",
     "value": 52.4,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1, SIMPLER (Visual Matching)",
     "quote": "RT-1 85.7 44.2 73.0 6.5 52.4",
     "percent": 52.4,
     "method_date": "2022-12-13",
     "method_url": "https://arxiv.org/abs/2212.06817",
     "plot_date": "2024-05-09",
     "note": "",
     "id": "google-robot-pro-rt-1-2024-11-29"
    },
    {
     "policy": "RT-2-X",
     "date": "2024-11-29",
     "value": 46.3,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1, SIMPLER (Visual Matching)",
     "quote": "RT-2-X 78.7 77.9 25.0 3.7 46.3",
     "percent": 46.3,
     "method_date": "2023-10-13",
     "method_url": "https://arxiv.org/abs/2310.08864",
     "plot_date": "2024-05-09",
     "note": "",
     "id": "google-robot-pro-rt-2-x-2024-11-29"
    },
    {
     "policy": "Octo-Base",
     "date": "2024-11-29",
     "value": 11.0,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1, SIMPLER (Visual Matching)",
     "quote": "Octo-Base 17.0 4.2 22.7 0.0 11.0",
     "percent": 11.0,
     "method_date": "2024-05-20",
     "method_url": "https://arxiv.org/abs/2405.12213",
     "plot_date": "2024-05-20",
     "note": "",
     "id": "google-robot-pro-octo-base-2024-11-29"
    },
    {
     "policy": "OpenVLA",
     "date": "2024-11-29",
     "value": 34.3,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1, SIMPLER (Visual Matching)",
     "quote": "OpenVLA 18.0 56.3 63.0 0.0 34.3",
     "percent": 34.3,
     "method_date": "2024-06-13",
     "method_url": "https://arxiv.org/abs/2406.09246",
     "plot_date": "2024-06-13",
     "note": "",
     "id": "google-robot-pro-openvla-2024-11-29"
    },
    {
     "policy": "CogACT",
     "date": "2024-11-29",
     "value": 74.8,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1, SIMPLER (Visual Matching)",
     "quote": "Ours (CogACT) 91.3 85.0 71.8 50.9 74.8",
     "percent": 74.8,
     "method_date": "2024-11-29",
     "method_url": "https://arxiv.org/abs/2411.19650",
     "plot_date": "2024-11-29",
     "note": "",
     "id": "google-robot-pro-cogact-2024-11-29"
    }
   ],
   "robustness_rows": [
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "Octo-Base",
     "clean_value": 11.0,
     "perturbed_value": 1.2,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "Octo-Base VM avg 11.0 | VA: Octo-Base 0.6 3.1 1.1 0.0 1.2",
     "clean_percent": 11.0,
     "perturbed_percent": 1.2,
     "drop_points": 9.8,
     "retained": 0.1091,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-octo-base"
    },
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "RT-1-X",
     "clean_value": 42.4,
     "perturbed_value": 30.2,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "RT-1-X VM avg 42.4 | VA: RT-1-X 49.0 32.3 29.4 10.1 30.2",
     "clean_percent": 42.4,
     "perturbed_percent": 30.2,
     "drop_points": 12.2,
     "retained": 0.7123,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-rt-1-x"
    },
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "CogACT",
     "clean_value": 74.8,
     "perturbed_value": 61.3,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "Ours VM avg 74.8 | VA: Ours 89.6 80.8 28.3 46.6 61.3",
     "clean_percent": 74.8,
     "perturbed_percent": 61.3,
     "drop_points": 13.5,
     "retained": 0.8195,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-cogact"
    },
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "RT-1",
     "clean_value": 52.4,
     "perturbed_value": 43.7,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "RT-1 VM avg 52.4 | VA: RT-1 89.8 50.0 32.3 2.6 43.7",
     "clean_percent": 52.4,
     "perturbed_percent": 43.7,
     "drop_points": 8.7,
     "retained": 0.834,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-rt-1"
    },
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "OpenVLA",
     "clean_value": 34.3,
     "perturbed_value": 39.3,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "OpenVLA VM avg 34.3 | VA: OpenVLA 60.8 67.7 28.8 0.0 39.3",
     "clean_percent": 34.3,
     "perturbed_percent": 39.3,
     "drop_points": -5.0,
     "retained": 1.1458,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-openvla"
    },
    {
     "perturbation_suite": "SIMPLER Variant Aggregation",
     "perturbation": "variant aggregation (backgrounds, lighting, distractors, table textures, camera poses)",
     "policy": "RT-2-X",
     "clean_value": 46.3,
     "perturbed_value": 54.4,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 1 (SIMPLER Visual Matching vs Variant Aggregation, Average column)",
     "quote": "RT-2-X VM avg 46.3 | VA: RT-2-X 82.3 79.2 35.3 20.6 54.4",
     "clean_percent": 46.3,
     "perturbed_percent": 54.4,
     "drop_points": -8.1,
     "retained": 1.1749,
     "note": "'Clean' here = Visual Matching. VA is not strictly harder: RT-2-X and OpenVLA score higher under VA.",
     "id": "google-robot-rob-variant-aggregation-backgrounds-lighting-distractors-table-textures-camera-poses-rt-2-x"
    }
   ],
   "fidelity_rows": [
    {
     "statistic": "pearson_r",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, Pearson r, SIMPLER-VisMatch, Average",
     "quote": "SIMPLER-VisMatch (ours) 0.976 0.855 0.942 0.924",
     "value": 0.924,
     "n_policies": 6,
     "note": "Visual Matching; 3-task average (Pick Coke Can, Move Near, Drawer)."
    },
    {
     "statistic": "pearson_r",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, Pearson r, SIMPLER-VarAgg, Average",
     "quote": "SIMPLER-VarAgg (ours) 0.960 0.887 0.486 0.778",
     "value": 0.778,
     "n_policies": 6,
     "note": "Variant Aggregation."
    },
    {
     "statistic": "pearson_r",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, Pearson r, Validation MSE, Average",
     "quote": "Validation MSE 0.464 0.230 0.231 0.308",
     "value": 0.308,
     "n_policies": 6,
     "note": "offline validation MSE (proxy). Baseline proxy (offline validation MSE), not simulation."
    },
    {
     "statistic": "mmrv",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, MMRV, Validation MSE, Average",
     "quote": "Validation MSE 0.412 0.408 0.306 0.375",
     "value": 0.375,
     "n_policies": 6,
     "note": "offline validation MSE (proxy). Baseline proxy."
    },
    {
     "statistic": "mmrv",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, MMRV, SIMPLER-VarAgg, Average",
     "quote": "SIMPLER-VarAgg (ours) 0.084 0.111 0.235 0.143",
     "value": 0.143,
     "n_policies": 6,
     "note": "Variant Aggregation."
    },
    {
     "statistic": "mmrv",
     "real_reference": "real Google Robot evals of 6 checkpoints (3 RT-1, RT-1-X, RT-2-X, Octo-Base)",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Table I, MMRV, SIMPLER-VisMatch, Average",
     "quote": "SIMPLER-VisMatch (ours) 0.031 0.111 0.027 0.056",
     "value": 0.056,
     "n_policies": 6,
     "note": "Visual Matching."
    },
    {
     "statistic": "statement",
     "real_reference": "real-world performance of recent generalist policies",
     "source_url": "https://arxiv.org/abs/2512.16881",
     "locator": "Sec 2 (Related Work)",
     "quote": "it fails to deliver strong real-world performance correlation for recent generalist policies.",
     "value": null,
     "n_policies": null,
     "note": "Qualitative statement about SIMPLER; no number given. Also Sec 5.2: 'Libero-score too is poorly correlated with real-world performance'."
    }
   ]
  },
  {
   "id": "sim--sim-manipulation-eval--meta-world",
   "name": "Meta-World",
   "metric": "MT50-rand success rate at 20M env steps",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#sim--sim-manipulation-eval--meta-world",
   "released": "2019-10-24",
   "note": "",
   "ceiling": null,
   "best_percent": 72.9,
   "headroom_percent": 27.1,
   "first_date": "2019-10-24",
   "latest_date": "2023-11-19",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2019-10-24",
     "percent": 49.3,
     "policy": "MTSAC",
     "id": "meta-world-pro-mtsac-2023-11-19"
    },
    {
     "date": "2021-02-11",
     "percent": 50.8,
     "policy": "CARE",
     "id": "meta-world-pro-care-2023-11-19"
    },
    {
     "date": "2022-10-21",
     "percent": 57.3,
     "policy": "PaCo",
     "id": "meta-world-pro-paco-2023-11-19"
    },
    {
     "date": "2023-11-19",
     "percent": 72.9,
     "policy": "MOORE",
     "id": "meta-world-pro-moore-2023-11-19"
    }
   ],
   "progress": [
    {
     "policy": "MTSAC",
     "date": "2023-11-19",
     "value": 49.3,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2311.11385",
     "locator": "Table 2 (MT50-rand)",
     "quote": "MTSAC (Yu et al., 2019) 49.3±1.5",
     "percent": 49.3,
     "method_date": "2019-10-24",
     "method_url": "https://arxiv.org/abs/1910.10897",
     "plot_date": "2019-10-24",
     "note": "Baselines taken by MOORE from Sun et al. (2022). Protocol differs from the original Meta-World Table 1; do not join the two series.",
     "id": "meta-world-pro-mtsac-2023-11-19"
    },
    {
     "policy": "CARE",
     "date": "2023-11-19",
     "value": 50.8,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2311.11385",
     "locator": "Table 2 (MT50-rand)",
     "quote": "CARE (Sodhani et al., 2021) 50.8±1.0",
     "percent": 50.8,
     "method_date": "2021-02-11",
     "method_url": "https://arxiv.org/abs/2102.06177",
     "plot_date": "2021-02-11",
     "note": "Baselines taken by MOORE from Sun et al. (2022). Protocol differs from the original Meta-World Table 1; do not join the two series.",
     "id": "meta-world-pro-care-2023-11-19"
    },
    {
     "policy": "PaCo",
     "date": "2023-11-19",
     "value": 57.3,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2311.11385",
     "locator": "Table 2 (MT50-rand)",
     "quote": "PaCo (Sun et al., 2022) 57.3±1.3",
     "percent": 57.3,
     "method_date": "2022-10-21",
     "method_url": "https://arxiv.org/abs/2210.11653",
     "plot_date": "2022-10-21",
     "note": "Baselines taken by MOORE from Sun et al. (2022). Protocol differs from the original Meta-World Table 1; do not join the two series.",
     "id": "meta-world-pro-paco-2023-11-19"
    },
    {
     "policy": "MOORE",
     "date": "2023-11-19",
     "value": 72.9,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2311.11385",
     "locator": "Table 2 (MT50-rand)",
     "quote": "MOORE (ours) 72.9±3.3",
     "percent": 72.9,
     "method_date": "2023-11-19",
     "method_url": "https://arxiv.org/abs/2311.11385",
     "plot_date": "2023-11-19",
     "note": "Baselines taken by MOORE from Sun et al. (2022). Protocol differs from the original Meta-World Table 1; do not join the two series.",
     "id": "meta-world-pro-moore-2023-11-19"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  },
  {
   "id": "vln--vision-language-navigation--r2r",
   "name": "Room-to-Room",
   "metric": "R2R-CE (VLN-CE, continuous) val-unseen success rate",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#vln--vision-language-navigation--r2r",
   "released": "2017-11-20",
   "note": "Human success on discrete R2R is 86.4% (https://arxiv.org/abs/1711.07280, Table 1); no R2R-CE human figure is published.",
   "ceiling": null,
   "best_percent": 61.7,
   "headroom_percent": 38.3,
   "first_date": "2020-04-06",
   "latest_date": "2025-09-15",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2020-04-06",
     "percent": 32.0,
     "policy": "CMA (RGB-D)",
     "id": "r2r-pro-cma-rgb-d-2025-09-15"
    },
    {
     "date": "2023-04-06",
     "percent": 57.0,
     "policy": "ETPNav",
     "id": "r2r-pro-etpnav-2025-09-15"
    },
    {
     "date": "2024-04-02",
     "percent": 61.0,
     "policy": "HNR",
     "id": "r2r-pro-hnr-2025-09-15"
    },
    {
     "date": "2025-09-15",
     "percent": 61.7,
     "policy": "NavFoM (four views)",
     "id": "r2r-pro-navfom-four-views-2025-09-15"
    }
   ],
   "progress": [
    {
     "policy": "CMA (RGB-D)",
     "date": "2025-09-15",
     "value": 32.0,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "CMA (Krantz et al., 2020) 7.37 40.0 32.0 30.0",
     "percent": 32.0,
     "method_date": "2020-04-06",
     "method_url": "https://arxiv.org/abs/2004.02857",
     "plot_date": "2020-04-06",
     "note": "Observation setting: single RGB + depth. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-cma-rgb-d-2025-09-15"
    },
    {
     "policy": "Seq2Seq (RGB-D)",
     "date": "2025-09-15",
     "value": 25.0,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "Seq2Seq (Krantz et al., 2020) 7.77 37.0 25.0 22.0",
     "percent": 25.0,
     "method_date": "2020-04-06",
     "method_url": "https://arxiv.org/abs/2004.02857",
     "plot_date": "2020-04-06",
     "note": "Observation setting: single RGB + depth. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-seq2seq-rgb-d-2025-09-15"
    },
    {
     "policy": "ETPNav",
     "date": "2025-09-15",
     "value": 57.0,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "ETPNav* (An et al., 2024) 4.71 65.0 57.0 49.0",
     "percent": 57.0,
     "method_date": "2023-04-06",
     "method_url": "https://arxiv.org/abs/2304.03047",
     "plot_date": "2023-04-06",
     "note": "Observation setting: panoramic RGB + depth + odometry. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-etpnav-2025-09-15"
    },
    {
     "policy": "NaVid",
     "date": "2025-09-15",
     "value": 41.9,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "NaVid (Zhang et al., 2024a) 5.72 49.2 41.9 36.5",
     "percent": 41.9,
     "method_date": "2024-02-24",
     "method_url": "https://arxiv.org/abs/2402.15852",
     "plot_date": "2024-02-24",
     "note": "Observation setting: single RGB only. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-navid-2025-09-15"
    },
    {
     "policy": "HNR",
     "date": "2025-09-15",
     "value": 61.0,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "HNR* (Wang et al., 2024b) 4.42 67.0 61.0 51.0",
     "percent": 61.0,
     "method_date": "2024-04-02",
     "method_url": "https://arxiv.org/abs/2404.01943",
     "plot_date": "2024-04-02",
     "note": "Observation setting: panoramic RGB + depth + odometry. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-hnr-2025-09-15"
    },
    {
     "policy": "StreamVLN (RGB-only)",
     "date": "2025-09-15",
     "value": 55.7,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "StreamVLN-RGB-only (Wei et al., 2025) 5.10 64.0 55.7 50.9",
     "percent": 55.7,
     "method_date": "2025-07-07",
     "method_url": "https://arxiv.org/abs/2507.05240",
     "plot_date": "2025-07-07",
     "note": "Observation setting: single RGB only. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-streamvln-rgb-only-2025-09-15"
    },
    {
     "policy": "NavFoM (four views)",
     "date": "2025-09-15",
     "value": 61.7,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "NavFoM (Four views) 4.61 72.1 61.7 55.3",
     "percent": 61.7,
     "method_date": "2025-09-15",
     "method_url": "https://arxiv.org/abs/2509.12129",
     "plot_date": "2025-09-15",
     "note": "Observation setting: four RGB views, no depth. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-navfom-four-views-2025-09-15"
    },
    {
     "policy": "NavFoM (single view)",
     "date": "2025-09-15",
     "value": 56.2,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2509.12129",
     "locator": "Table 1 (Comparison on VLN-CE in Single-View and Multi-View Settings), R2R Val-Unseen",
     "quote": "NavFoM (Single view) 5.01 64.9 56.2 51.2",
     "percent": 56.2,
     "method_date": "2025-09-15",
     "method_url": "https://arxiv.org/abs/2509.12129",
     "plot_date": "2025-09-15",
     "note": "Observation setting: single RGB only. Columns NE/OS/SR/SPL.",
     "id": "r2r-pro-navfom-single-view-2025-09-15"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  },
  {
   "id": "alfred--household-instruction-sim--alfred",
   "name": "ALFRED",
   "metric": "task success rate, test unseen",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#alfred--household-instruction-sim--alfred",
   "released": "2019-12-03",
   "note": "",
   "ceiling": {
    "label": "Human",
    "percent": 91.0,
    "source_url": "https://arxiv.org/abs/1912.01734",
    "locator": "Table 3",
    "quote": "HUMAN - 91.0 (85.8)"
   },
   "best_percent": 60.79,
   "headroom_percent": 39.21,
   "first_date": "2019-12-03",
   "latest_date": "2026-04-15",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2019-12-03",
     "percent": 0.5,
     "policy": "Seq2Seq",
     "id": "alfred-pro-seq2seq-2019-12-03"
    },
    {
     "date": "2021-10-12",
     "percent": 26.49,
     "policy": "FILM (step-by-step instr.)",
     "id": "alfred-pro-film-step-by-step-instr-2026-04-15"
    },
    {
     "date": "2022-11-07",
     "percent": 45.32,
     "policy": "Prompter (step-by-step instr.)",
     "id": "alfred-pro-prompter-step-by-step-instr-2026-04-15"
    },
    {
     "date": "2023-08-14",
     "percent": 46.11,
     "policy": "CAPEAM (step-by-step instr.)",
     "id": "alfred-pro-capeam-step-by-step-instr-2026-04-15"
    },
    {
     "date": "2023-12-12",
     "percent": 57.82,
     "policy": "ThinkBot (step-by-step instr.)",
     "id": "alfred-pro-thinkbot-step-by-step-instr-2026-04-15"
    },
    {
     "date": "2026-04-15",
     "percent": 60.79,
     "policy": "ESCAPE (step-by-step instr.)",
     "id": "alfred-pro-escape-step-by-step-instr-2026-04-15"
    }
   ],
   "progress": [
    {
     "policy": "Seq2Seq",
     "date": "2019-12-03",
     "value": 0.5,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/1912.01734",
     "locator": "Table 3 (Task and Goal-Condition Success), Test",
     "quote": "SEQ2SEQ 2.1 (1.0) 0.5 (0.2)",
     "percent": 0.5,
     "method_date": "2019-12-03",
     "method_url": "https://arxiv.org/abs/1912.01734",
     "plot_date": "2019-12-03",
     "note": "",
     "id": "alfred-pro-seq2seq-2019-12-03"
    },
    {
     "policy": "FILM (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 26.49,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "FILM ✓ 27.67 38.51 11.23 15.06 26.49 36.37 10.55 14.30",
     "percent": 26.49,
     "method_date": "2021-10-12",
     "method_url": "https://arxiv.org/abs/2110.07342",
     "plot_date": "2021-10-12",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-film-step-by-step-instr-2026-04-15"
    },
    {
     "policy": "Prompter (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 45.32,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "Prompter ✓ 51.17 60.22 25.12 30.21 45.32 56.57 20.79 25.80",
     "percent": 45.32,
     "method_date": "2022-11-07",
     "method_url": "https://arxiv.org/abs/2211.03267",
     "plot_date": "2022-11-07",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-prompter-step-by-step-instr-2026-04-15"
    },
    {
     "policy": "CAPEAM (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 46.11,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "CAPEAM ✓ 51.79 60.50 21.60 25.88 46.11 57.33 19.45 24.06",
     "percent": 46.11,
     "method_date": "2023-08-14",
     "method_url": "https://arxiv.org/abs/2308.07241",
     "plot_date": "2023-08-14",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-capeam-step-by-step-instr-2026-04-15"
    },
    {
     "policy": "ThinkBot (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 57.82,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "ThinkBot ✓ 62.69 71.64 32.02 37.01 57.82 67.75 26.93 30.73",
     "percent": 57.82,
     "method_date": "2023-12-12",
     "method_url": "https://arxiv.org/abs/2312.07062",
     "plot_date": "2023-12-12",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-thinkbot-step-by-step-instr-2026-04-15"
    },
    {
     "policy": "DISCO (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 56.5,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "DISCO ✓ 59.50 66.10 40.60 47.40 56.50 66.80 36.50 44.50",
     "percent": 56.5,
     "method_date": "2024-07-20",
     "method_url": "https://arxiv.org/abs/2407.14758",
     "plot_date": "2024-07-20",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-disco-step-by-step-instr-2026-04-15"
    },
    {
     "policy": "ESCAPE (step-by-step instr.)",
     "date": "2026-04-15",
     "value": 60.79,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2604.13633",
     "locator": "Table 1 (Performance comparison on ALFRED benchmark), Unseen SR",
     "quote": "ESCAPE (Ours) ✓ 65.09 72.88 52.42 58.29 60.79 68.60 46.82 52.57",
     "percent": 60.79,
     "method_date": null,
     "method_url": null,
     "plot_date": "2026-04-15",
     "note": "Baseline rows copied by ESCAPE authors; the official ALFRED leaderboard (leaderboard.allenai.org) blocks robots and was not checked.",
     "id": "alfred-pro-escape-step-by-step-instr-2026-04-15"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  },
  {
   "id": "vla-manipulation--vla-manipulation-eval--robotwin-2",
   "name": "RoboTwin 2.0",
   "metric": "average success, Easy (clean) setting",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#vla-manipulation--vla-manipulation-eval--robotwin-2",
   "released": "2025-06-22",
   "note": "",
   "ceiling": null,
   "best_percent": 55.2,
   "headroom_percent": 44.8,
   "first_date": "2025-06-22",
   "latest_date": "2025-06-22",
   "median_retained": 0.0906,
   "worst_retained": 0.0214,
   "frontier": [
    {
     "date": "2025-06-22",
     "percent": 55.2,
     "policy": "DP3",
     "id": "robotwin-2-pro-dp3-2025-06-22"
    }
   ],
   "progress": [
    {
     "policy": "DP3",
     "date": "2025-06-22",
     "value": 55.2,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "DP3 Easy 55.2% / Hard 5.0%",
     "percent": 55.2,
     "method_date": "2024-03-06",
     "method_url": "https://arxiv.org/abs/2403.03954",
     "plot_date": "2025-06-22",
     "note": "Subset average (see caption).",
     "id": "robotwin-2-pro-dp3-2025-06-22"
    },
    {
     "policy": "Pi0",
     "date": "2025-06-22",
     "value": 46.4,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. Full results in Appendix K and Leaderboard. | Pi0 Easy 46.4%",
     "percent": 46.4,
     "method_date": "2024-10-31",
     "method_url": "https://arxiv.org/abs/2410.24164",
     "plot_date": "2025-06-22",
     "note": "Caption says subset; average may not be over all 50 tasks. X-VLA's 'Maximum of Existing SOTA' row also prints 46.4.",
     "id": "robotwin-2-pro-pi0-2025-06-22"
    }
   ],
   "robustness_rows": [
    {
     "perturbation_suite": "RoboTwin 2.0 Hard (domain randomization)",
     "perturbation": "clutter, lighting, background, texture, table height randomization (Hard setting)",
     "policy": "DP",
     "clean_value": 28.0,
     "perturbed_value": 0.6,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. | DP Easy 28.0% Hard 0.6%",
     "clean_percent": 28.0,
     "perturbed_percent": 0.6,
     "drop_points": 27.4,
     "retained": 0.0214,
     "note": "Average over the subset of tasks displayed in Table 5, not necessarily all 50.",
     "id": "robotwin-2-rob-clutter-lighting-background-texture-table-height-randomization-hard-setting-dp"
    },
    {
     "perturbation_suite": "RoboTwin 2.0 Hard (domain randomization)",
     "perturbation": "clutter, lighting, background, texture, table height randomization (Hard setting)",
     "policy": "ACT",
     "clean_value": 29.7,
     "perturbed_value": 1.7,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. | ACT Easy 29.7% Hard 1.7%",
     "clean_percent": 29.7,
     "perturbed_percent": 1.7,
     "drop_points": 28.0,
     "retained": 0.0572,
     "note": "Average over the subset of tasks displayed in Table 5, not necessarily all 50.",
     "id": "robotwin-2-rob-clutter-lighting-background-texture-table-height-randomization-hard-setting-act"
    },
    {
     "perturbation_suite": "RoboTwin 2.0 Hard (domain randomization)",
     "perturbation": "clutter, lighting, background, texture, table height randomization (Hard setting)",
     "policy": "DP3",
     "clean_value": 55.2,
     "perturbed_value": 5.0,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. | DP3 Easy 55.2% Hard 5.0%",
     "clean_percent": 55.2,
     "perturbed_percent": 5.0,
     "drop_points": 50.2,
     "retained": 0.0906,
     "note": "Average over the subset of tasks displayed in Table 5, not necessarily all 50.",
     "id": "robotwin-2-rob-clutter-lighting-background-texture-table-height-randomization-hard-setting-dp3"
    },
    {
     "perturbation_suite": "RoboTwin 2.0 Hard (domain randomization)",
     "perturbation": "clutter, lighting, background, texture, table height randomization (Hard setting)",
     "policy": "Pi0",
     "clean_value": 46.4,
     "perturbed_value": 16.3,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. | Pi0 Easy 46.4% Hard 16.3%",
     "clean_percent": 46.4,
     "perturbed_percent": 16.3,
     "drop_points": 30.1,
     "retained": 0.3513,
     "note": "Average over the subset of tasks displayed in Table 5, not necessarily all 50.",
     "id": "robotwin-2-rob-clutter-lighting-background-texture-table-height-randomization-hard-setting-pi0"
    },
    {
     "perturbation_suite": "RoboTwin 2.0 Hard (domain randomization)",
     "perturbation": "clutter, lighting, background, texture, table height randomization (Hard setting)",
     "policy": "RDT",
     "clean_value": 34.5,
     "perturbed_value": 13.7,
     "source_url": "https://arxiv.org/abs/2506.18088",
     "locator": "Table 5 (average row)",
     "quote": "Subset of RoboTwin 2.0 benchmark. | RDT Easy 34.5% Hard 13.7%",
     "clean_percent": 34.5,
     "perturbed_percent": 13.7,
     "drop_points": 20.8,
     "retained": 0.3971,
     "note": "Average over the subset of tasks displayed in Table 5, not necessarily all 50.",
     "id": "robotwin-2-rob-clutter-lighting-background-texture-table-height-randomization-hard-setting-rdt"
    }
   ],
   "fidelity_rows": []
  },
  {
   "id": "simpler--real-to-sim-eval--widowx",
   "name": "WidowX",
   "metric": "average success, 4 WidowX/Bridge tasks (spoon on towel, carrot on plate, stack block, eggplant in basket), Visual Matching",
   "unit": "percent",
   "scale_max": 100,
   "medium": "paired",
   "catalogue_href": "/topics/benchmarks.html#simpler--real-to-sim-eval--widowx",
   "released": "2024-05-09",
   "note": "",
   "ceiling": null,
   "best_percent": 51.3,
   "headroom_percent": 48.7,
   "first_date": "2024-05-09",
   "latest_date": "2024-11-29",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2024-05-09",
     "percent": 1.1,
     "policy": "RT-1-X",
     "id": "widowx-pro-rt-1-x-2024-11-29"
    },
    {
     "date": "2024-05-20",
     "percent": 26.7,
     "policy": "Octo-Small",
     "id": "widowx-pro-octo-small-2024-11-29"
    },
    {
     "date": "2024-11-29",
     "percent": 51.3,
     "policy": "CogACT",
     "id": "widowx-pro-cogact-2024-11-29"
    }
   ],
   "progress": [
    {
     "policy": "RT-1-X",
     "date": "2024-11-29",
     "value": 1.1,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 2 (WidowX, SIMPLER Visual Matching)",
     "quote": "RT-1-X 0.0 4.2 0.0 0.0 1.1",
     "percent": 1.1,
     "method_date": "2023-10-13",
     "method_url": "https://arxiv.org/abs/2310.08864",
     "plot_date": "2024-05-09",
     "note": "",
     "id": "widowx-pro-rt-1-x-2024-11-29"
    },
    {
     "policy": "Octo-Small",
     "date": "2024-11-29",
     "value": 26.7,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 2 (WidowX, SIMPLER Visual Matching)",
     "quote": "Octo-Small 41.7 8.2 0.0 56.7 26.7",
     "percent": 26.7,
     "method_date": "2024-05-20",
     "method_url": "https://arxiv.org/abs/2405.12213",
     "plot_date": "2024-05-20",
     "note": "",
     "id": "widowx-pro-octo-small-2024-11-29"
    },
    {
     "policy": "OpenVLA",
     "date": "2024-11-29",
     "value": 4.2,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 2 (WidowX, SIMPLER Visual Matching)",
     "quote": "OpenVLA 4.2 0.0 0.0 12.5 4.2",
     "percent": 4.2,
     "method_date": "2024-06-13",
     "method_url": "https://arxiv.org/abs/2406.09246",
     "plot_date": "2024-06-13",
     "note": "",
     "id": "widowx-pro-openvla-2024-11-29"
    },
    {
     "policy": "CogACT",
     "date": "2024-11-29",
     "value": 51.3,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2411.19650",
     "locator": "Table 2 (WidowX, SIMPLER Visual Matching)",
     "quote": "Ours (CogACT) 71.7 50.8 15.0 67.5 51.3",
     "percent": 51.3,
     "method_date": "2024-11-29",
     "method_url": "https://arxiv.org/abs/2411.19650",
     "plot_date": "2024-11-29",
     "note": "",
     "id": "widowx-pro-cogact-2024-11-29"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": [
    {
     "statistic": "statement",
     "real_reference": "real WidowX + BridgeData V2 evaluations of RT-1-X, Octo-Base and Octo-Small",
     "source_url": "https://arxiv.org/abs/2405.05941",
     "locator": "Fig. 6 caption",
     "quote": "SIMPLER evaluations have strong correlation with real policy performance.",
     "value": null,
     "n_policies": 3,
     "note": "The WidowX correlation values appear only in a figure and are not copied here."
    }
   ]
  },
  {
   "id": "sim--sim-manipulation-eval--robocasa",
   "name": "RoboCasa",
   "metric": "average success, RoboCasa, 100 demos per task",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#sim--sim-manipulation-eval--robocasa",
   "released": "2024-06-04",
   "note": "",
   "ceiling": null,
   "best_percent": 32.1,
   "headroom_percent": 67.9,
   "first_date": "2024-06-04",
   "latest_date": "2025-03-18",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2024-06-04",
     "percent": 25.6,
     "policy": "Diffusion Policy",
     "id": "robocasa-pro-diffusion-policy-2025-03-18"
    },
    {
     "date": "2025-03-18",
     "percent": 32.1,
     "policy": "GR00T-N1-2B",
     "id": "robocasa-pro-gr00t-n1-2b-2025-03-18"
    }
   ],
   "progress": [
    {
     "policy": "Diffusion Policy",
     "date": "2025-03-18",
     "value": 25.6,
     "self_reported": false,
     "source_url": "https://arxiv.org/abs/2503.14734",
     "locator": "Table 2 (Simulation Results)",
     "quote": "Diffusion Policy 25.6% 56.1% 32.7% 33.4% (RoboCasa | DexMG | GR-1 | Average)",
     "percent": 25.6,
     "method_date": "2023-03-07",
     "method_url": "https://arxiv.org/abs/2303.04137",
     "plot_date": "2024-06-04",
     "note": "",
     "id": "robocasa-pro-diffusion-policy-2025-03-18"
    },
    {
     "policy": "GR00T-N1-2B",
     "date": "2025-03-18",
     "value": 32.1,
     "self_reported": true,
     "source_url": "https://arxiv.org/abs/2503.14734",
     "locator": "Table 2",
     "quote": "GR00T-N1-2B 32.1% 66.5% 50.0% 45.0%",
     "percent": 32.1,
     "method_date": "2025-03-18",
     "method_url": "https://arxiv.org/abs/2503.14734",
     "plot_date": "2025-03-18",
     "note": "100-demo regime; not comparable with the 50-human / 3000-generated regimes of the RoboCasa paper.",
     "id": "robocasa-pro-gr00t-n1-2b-2025-03-18"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  },
  {
   "id": "activity--everyday-activity-sim--behavior-1k",
   "name": "BEHAVIOR-1K",
   "metric": "BEHAVIOR Challenge 2025 full task success rate, held-out test",
   "unit": "percent",
   "scale_max": 100,
   "medium": "sim",
   "catalogue_href": "/topics/benchmarks.html#activity--everyday-activity-sim--behavior-1k",
   "released": "2024-03-14",
   "note": "",
   "ceiling": null,
   "best_percent": 12.4,
   "headroom_percent": 87.6,
   "first_date": "2025-11-14",
   "latest_date": "2025-11-14",
   "median_retained": null,
   "worst_retained": null,
   "frontier": [
    {
     "date": "2025-11-14",
     "percent": 12.4,
     "policy": "Robot Learning Collective",
     "id": "behavior-1k-pro-robot-learning-collective-2025-11-14"
    }
   ],
   "progress": [
    {
     "policy": "Robot Learning Collective",
     "date": "2025-11-14",
     "value": 12.4,
     "self_reported": false,
     "source_url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
     "locator": "Standard track, rank 1",
     "quote": "Full Task Success (Public Val / Held-out) 0.1120 / 0.1240",
     "percent": 12.4,
     "method_date": null,
     "method_url": null,
     "plot_date": "2025-11-14",
     "note": "",
     "id": "behavior-1k-pro-robot-learning-collective-2025-11-14"
    }
   ],
   "robustness_rows": [],
   "fidelity_rows": []
  }
 ],
 "proxies": [
  {
   "suite_name": "Ctrl-World (video model) as evaluated in PolaRiS",
   "statistic": "mmrv",
   "real_reference": "real DROID evaluations",
   "source_url": "https://arxiv.org/abs/2512.16881",
   "locator": "Sec 5.2",
   "quote": "leading to clear policy mis-rankings (MMRV: 0.22)",
   "value": 0.22,
   "n_policies": 4,
   "note": "Baseline: Ctrl-World video world model, reported in the PolaRiS paper."
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (T-block pushing)",
   "statistic": "pearson_r",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I (MMRV ↓ / r ↑), Ours, T-block pushing",
   "quote": "Ours ... T-block pushing: 0.108 / 0.915",
   "value": 0.915,
   "n_policies": 4,
   "note": "n_policies counts architectures; multiple checkpoints each."
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (T-block pushing)",
   "statistic": "mmrv",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I, Ours, T-block pushing",
   "quote": "Ours ... T-block pushing: 0.108 / 0.915",
   "value": 0.108,
   "n_policies": 4,
   "note": ""
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (rope routing)",
   "statistic": "pearson_r",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I (MMRV ↓ / r ↑), Ours, rope routing",
   "quote": "Ours ... rope routing: 0.174 / 0.901",
   "value": 0.901,
   "n_policies": 4,
   "note": "n_policies counts architectures; multiple checkpoints each."
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (rope routing)",
   "statistic": "mmrv",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I, Ours, rope routing",
   "quote": "Ours ... rope routing: 0.174 / 0.901",
   "value": 0.174,
   "n_policies": 4,
   "note": ""
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (toy packing)",
   "statistic": "pearson_r",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I (MMRV ↓ / r ↑), Ours, toy packing",
   "quote": "Ours ... toy packing: 0.076 / 0.944",
   "value": 0.944,
   "n_policies": 4,
   "note": "n_policies counts architectures; multiple checkpoints each."
  },
  {
   "suite_name": "Gaussian-splatting soft-body real-to-sim (toy packing)",
   "statistic": "mmrv",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Table I, Ours, toy packing",
   "quote": "Ours ... toy packing: 0.076 / 0.944",
   "value": 0.076,
   "n_policies": 4,
   "note": ""
  },
  {
   "suite_name": "IsaacLab baseline (push-T)",
   "statistic": "pearson_r",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Sec IV-C; Table I IsaacLab",
   "quote": "the baseline yields only r=0.649 on push-T, and an even lower r=0.237 on rope routing",
   "value": 0.649,
   "n_policies": 4,
   "note": "IsaacLab baseline, push-T."
  },
  {
   "suite_name": "IsaacLab baseline (rope routing)",
   "statistic": "pearson_r",
   "real_reference": "real-robot evals of ACT, DP, Pi-0, SmolVLA checkpoints (16–27 episodes per checkpoint)",
   "source_url": "https://arxiv.org/abs/2511.04665",
   "locator": "Sec IV-C; Table I IsaacLab",
   "quote": "the baseline yields only r=0.649 on push-T, and an even lower r=0.237 on rope routing",
   "value": 0.237,
   "n_policies": 4,
   "note": "IsaacLab baseline, rope routing."
  },
  {
   "suite_name": "PolaRiS",
   "statistic": "pearson_r",
   "real_reference": "RoboArena policy scores",
   "source_url": "https://arxiv.org/abs/2512.16881",
   "locator": "Sec 1; Fig. 8",
   "quote": "PolaRiS evaluations also strongly correlate (r=0.98) with policy scores in RoboArena",
   "value": 0.98,
   "n_policies": null,
   "note": "Reference is RoboArena (distributed real-world pairwise eval), not direct rollouts."
  },
  {
   "suite_name": "PolaRiS",
   "statistic": "pearson_r",
   "real_reference": "real DROID evaluations, 6 paired environments, 20 rollouts per policy per env",
   "source_url": "https://arxiv.org/abs/2512.16881",
   "locator": "Sec 1 (Introduction); setup Sec 5.1",
   "quote": "with an average Pearson correlation of r=0.9",
   "value": 0.9,
   "n_policies": 4,
   "note": "Policies: π0, π0-FAST, PaliGemma-binning, π0.5."
  },
  {
   "suite_name": "PolaRiS",
   "statistic": "pearson_r",
   "real_reference": "real DROID evaluations",
   "source_url": "https://arxiv.org/abs/2512.16881",
   "locator": "Sec 5.2",
   "quote": "the worst-case correlation across all six tested environments is r=0.81",
   "value": 0.81,
   "n_policies": 4,
   "note": "Worst-case environment."
  },
  {
   "suite_name": "SimFoundry",
   "statistic": "pearson_r",
   "real_reference": "real-world evals on 7 manipulation tasks",
   "source_url": "https://arxiv.org/abs/2606.28276",
   "locator": "Abstract",
   "quote": "with mean Pearson correlation 0.911",
   "value": 0.911,
   "n_policies": 5,
   "note": "5 = policy architectures (per abstract); v4 2026-08-05."
  },
  {
   "suite_name": "SimFoundry",
   "statistic": "mmrv",
   "real_reference": "real-world evals on 7 manipulation tasks",
   "source_url": "https://arxiv.org/abs/2606.28276",
   "locator": "Abstract",
   "quote": "mean maximum ranking violation 0.018",
   "value": 0.018,
   "n_policies": 5,
   "note": ""
  }
 ]
}
