{
 "generated": "2026-10-10",
 "taxonomy": {
  "version": "v1-proposed",
  "levels": {
   "verified": {
    "label": "Verified",
    "short": "Checked against a primary source.",
    "long": "Confirmed on a primary source (the paper, official site, official repository, official leaderboard or official blog) on the date shown."
   },
   "reported": {
    "label": "Reported",
    "short": "Only found in a secondary source.",
    "long": "Only found in a secondary source such as news, a survey by other authors, or a third-party blog. Linked, not independently confirmed."
   },
   "inferred": {
    "label": "Inferred",
    "short": "Worked out by us from primary material.",
    "long": "Derived by the Atlas from primary material: a count, a calculation, or a reading of a licence. The note says how."
   },
   "unknown": {
    "label": "Unknown",
    "short": "Looked for, not found.",
    "long": "We looked in the primary sources listed and did not find it. We do not guess."
   }
  },
  "facets": {
   "kind": {
    "label": "What it is",
    "values": {
     "benchmark": {
      "label": "Benchmark",
      "def": "A fixed set of tasks, rules and scoring that anyone can run."
     },
     "dataset": {
      "label": "Dataset",
      "def": "Recorded robot data. It has no scoring rules of its own."
     },
     "platform": {
      "label": "Simulator",
      "def": "Software that hosts many tasks or benchmarks."
     },
     "arena": {
      "label": "Evaluation service",
      "def": "An organiser runs the policies that people send in and publishes the results."
     },
     "challenge": {
      "label": "Competition",
      "def": "A contest with fixed rules and a deadline."
     },
     "standard": {
      "label": "Standard",
      "def": "A formal test method published by a standards body."
     },
     "study": {
      "label": "In-house test",
      "def": "A test that only its authors run."
     }
    }
   },
   "venue": {
    "label": "Runs in",
    "order": [
     "offline",
     "sim",
     "world-model",
     "sim+real",
     "real"
    ],
    "values": {
     "offline": {
      "label": "Recorded data",
      "short": "Recorded",
      "def": "No robot moves. The system answers questions about recorded video or predicts what happens next."
     },
     "sim": {
      "label": "Simulation",
      "short": "Sim",
      "def": "A virtual robot carries out the tasks in a scene computed by a physics simulator."
     },
     "world-model": {
      "label": "AI simulator",
      "short": "AI sim",
      "def": "A trained neural network predicts what the robot would see after each action. It takes the place of a physics simulator."
     },
     "sim+real": {
      "label": "Sim + real",
      "short": "Sim + real",
      "def": "The benchmark has both simulated tests and real-robot tests."
     },
     "real": {
      "label": "Real robots",
      "short": "Real",
      "def": "Physical robots carry out the tasks."
     }
    }
   },
   "capability": {
    "label": "Tests",
    "values": {
     "manipulation": {
      "label": "Manipulation",
      "def": "Grasping, moving, placing and using objects."
     },
     "dexterous": {
      "label": "Dexterous hands",
      "def": "Multi-finger hands and in-hand manipulation."
     },
     "bimanual": {
      "label": "Two-arm tasks",
      "def": "Tasks that need two arms working together."
     },
     "mobile-manipulation": {
      "label": "Mobile manipulation",
      "def": "Moving the base and handling objects in one task."
     },
     "navigation": {
      "label": "Navigation",
      "def": "Reaching places or objects in a space."
     },
     "instruction-following": {
      "label": "Following instructions",
      "def": "Natural-language instructions drive the task."
     },
     "long-horizon": {
      "label": "Long tasks",
      "def": "Multi-step activities such as household chores."
     },
     "locomotion": {
      "label": "Locomotion",
      "def": "Walking, balance and whole-body control."
     },
     "collaboration": {
      "label": "Working with people",
      "def": "Collaboration with humans or other agents."
     },
     "embodied-reasoning": {
      "label": "Embodied reasoning",
      "def": "Spatial, physical or task reasoning without low-level control."
     },
     "world-modeling": {
      "label": "World models",
      "def": "Quality of predicted video or world states for robots."
     },
     "safety": {
      "label": "Safety",
      "def": "Avoiding harm and recognising unsafe or impossible requests."
     },
     "human-likeness": {
      "label": "Human-likeness",
      "def": "How human-like the motion or behaviour is."
     }
    }
   },
   "generalisation": {
    "label": "Varies at test time",
    "values": {
     "object-instance": {
      "label": "New objects",
      "def": "Objects not seen in training."
     },
     "object-pose": {
      "label": "Object positions",
      "def": "Objects start in different places."
     },
     "scene-layout": {
      "label": "Scene layout",
      "def": "Different rooms, layouts or furniture."
     },
     "visual": {
      "label": "Visual changes",
      "def": "Lighting, textures, backgrounds or camera views."
     },
     "language": {
      "label": "Wording",
      "def": "Different or new phrasings of instructions."
     },
     "new-task": {
      "label": "New tasks",
      "def": "Tasks or combinations not seen in training."
     },
     "embodiment": {
      "label": "New robots",
      "def": "A robot body not seen in training."
     },
     "none-stated": {
      "label": "None stated",
      "def": "The benchmark does not describe test-time variation."
     }
    }
   },
   "embodiment": {
    "label": "Robot",
    "values": {
     "single-arm": {
      "label": "One arm",
      "def": "One robot arm, usually fixed to a table."
     },
     "bimanual-arm": {
      "label": "Two arms",
      "def": "Two arms on a fixed base."
     },
     "mobile-manipulator": {
      "label": "Arm on wheels",
      "def": "An arm on a moving base."
     },
     "humanoid": {
      "label": "Humanoid",
      "def": "A human-shaped robot."
     },
     "legged": {
      "label": "Legged",
      "def": "Quadrupeds and other legged robots."
     },
     "dexterous-hand": {
      "label": "Robot hand",
      "def": "A multi-finger robot hand."
     },
     "mobile-base": {
      "label": "Wheels only",
      "def": "A navigating agent without an arm."
     },
     "cross-embodiment": {
      "label": "Many types",
      "def": "Data or tasks across many different robots."
     },
     "virtual-agent": {
      "label": "Game character",
      "def": "An avatar or agent in a game world."
     },
     "none": {
      "label": "No body",
      "def": "Video or questions only."
     }
    }
   },
   "scene": {
    "label": "Setting",
    "values": {
     "tabletop": {
      "label": "Tabletop"
     },
     "kitchen": {
      "label": "Kitchen"
     },
     "home": {
      "label": "Whole home"
     },
     "office-lab": {
      "label": "Office or lab"
     },
     "retail-logistics": {
      "label": "Retail or logistics"
     },
     "industrial": {
      "label": "Industrial"
     },
     "outdoor": {
      "label": "Outdoor"
     },
     "game-world": {
      "label": "Game world"
     },
     "mixed": {
      "label": "Mixed"
     }
    }
   },
   "scoring": {
    "label": "Scored by",
    "values": {
     "success-rate": {
      "label": "Success rate",
      "def": "Share of episodes where the task was completed."
     },
     "progress": {
      "label": "Progress score",
      "def": "Partial credit for partly completed tasks."
     },
     "chain-length": {
      "label": "Chain length",
      "def": "How many chained sub-tasks are completed in a row."
     },
     "path-efficiency": {
      "label": "Path efficiency",
      "def": "Success weighted by how direct the path was (SPL)."
     },
     "reward": {
      "label": "Reward",
      "def": "Task reward or return from the environment."
     },
     "preference": {
      "label": "Human preference",
      "def": "People compare two policies; results become a ranking."
     },
     "human-rating": {
      "label": "Human rating",
      "def": "People rate quality on a scale."
     },
     "auto-judge": {
      "label": "Automatic judge",
      "def": "A trained classifier or vision-language model judges the outcome."
     },
     "fidelity": {
      "label": "Fidelity",
      "def": "How closely generated video or motion matches a reference."
     },
     "accuracy": {
      "label": "Accuracy",
      "def": "Share of questions answered correctly."
     },
     "composite": {
      "label": "Composite index",
      "def": "A weighted mix of several metrics."
     }
    }
   },
   "leaderboard": {
    "label": "Leaderboard",
    "values": {
     "official": {
      "label": "Official leaderboard",
      "def": "A maintained public results page."
     },
     "paper-only": {
      "label": "None. Scores are only in papers.",
      "def": "Results live only in papers."
     },
     "community": {
      "label": "Run by the community",
      "def": "Maintained by people other than the authors."
     },
     "none": {
      "label": "None",
      "def": "No public results table."
     }
    }
   },
   "access": {
    "label": "Access",
    "values": {
     "open": {
      "label": "Open download"
     },
     "registration": {
      "label": "Download after registering"
     },
     "application": {
      "label": "By application"
     },
     "closed": {
      "label": "Closed"
     }
    }
   },
   "commercial_use": {
    "label": "Commercial use",
    "values": {
     "allowed": {
      "label": "Allowed",
      "def": "The licences shown allow commercial use. Our reading, not legal advice."
     },
     "non-commercial": {
      "label": "Not allowed",
      "def": "At least one part is licensed for non-commercial use only."
     },
     "unclear": {
      "label": "Unclear",
      "def": "Licences are missing, custom or conflicting."
     }
    }
   },
   "sim_to_real": {
    "label": "Checked against real robots",
    "ladder": true,
    "values": {
     "none-found": {
      "label": "Not checked",
      "rank": 0,
      "def": "No comparison with real robots has been published."
     },
     "claimed": {
      "label": "Claimed, not measured",
      "rank": 1,
      "def": "The authors say results carry over to real robots, but they published no paired comparison."
     },
     "demonstrated": {
      "label": "Tried on real robots, not compared",
      "rank": 2,
      "def": "Some policies were also run on real robots, but no comparison number was published."
     },
     "correlated": {
      "label": "Checked",
      "rank": 3,
      "def": "The same policies were scored in both settings, and a number compares the two."
     },
     "replicated": {
      "label": "Checked by others too",
      "rank": 4,
      "def": "An independent group also compared the two."
     },
     "not-applicable": {
      "label": "Real robots",
      "rank": null,
      "def": "The scores already come from real robots."
     }
    }
   },
   "status": {
    "label": "Status",
    "values": {
     "active": {
      "label": "Active",
      "def": "Updated in the last six months."
     },
     "maintained": {
      "label": "Maintained",
      "def": "Occasional fixes, no new versions recently."
     },
     "dormant": {
      "label": "Dormant",
      "def": "No updates for over a year."
     },
     "superseded": {
      "label": "Superseded",
      "def": "Replaced by a newer benchmark or version."
     }
    }
   },
   "builder_type": {
    "label": "Built by",
    "values": {
     "academic": {
      "label": "University lab"
     },
     "frontier-lab": {
      "label": "AI lab"
     },
     "robot-company": {
      "label": "Robot company"
     },
     "platform-vendor": {
      "label": "Platform vendor"
     },
     "consortium": {
      "label": "Consortium"
     },
     "standards-body": {
      "label": "Standards body"
     },
     "community": {
      "label": "Community"
     }
    }
   },
   "region": {
    "label": "Region",
    "values": {
     "north-america": {
      "label": "North America"
     },
     "europe": {
      "label": "Europe"
     },
     "china": {
      "label": "China"
     },
     "asia-other": {
      "label": "Asia (other)"
     },
     "multi": {
      "label": "Multi-region"
     }
    }
   },
   "issue": {
    "label": "Issue",
    "values": {
     "saturated": {
      "label": "Saturated",
      "def": "Top scores are near the ceiling, so it no longer separates strong systems."
     },
     "shortcut": {
      "label": "Shortcut",
      "def": "It can be solved without the skill it claims to test."
     },
     "protocol-variance": {
      "label": "Protocol variance",
      "def": "Results change with seeds, poses or settings."
     },
     "inconsistent-reporting": {
      "label": "Inconsistent reporting",
      "def": "Papers run it in different ways, so numbers do not line up."
     },
     "contamination": {
      "label": "Contamination",
      "def": "Test data may have leaked into training."
     },
     "other": {
      "label": "Other"
     }
    }
   },
   "uncertainty_reported": {
    "label": "Error bars",
    "values": {
     "yes": {
      "label": "Usually reported",
      "def": "Results normally come with error bars or confidence intervals."
     },
     "sometimes": {
      "label": "Sometimes reported",
      "def": "Some papers report uncertainty; many do not."
     },
     "no": {
      "label": "Not reported",
      "def": "Results are single numbers with no measure of uncertainty."
     }
    }
   },
   "evaluator": {
    "label": "Who runs it",
    "values": {
     "self-reported": {
      "label": "Each team tests its own model",
      "def": "Teams run the evaluation themselves and report their own numbers."
     },
     "organiser-run": {
      "label": "The organisers run the tests",
      "def": "The organisers run every submitted policy under the same conditions."
     },
     "both": {
      "label": "Both teams and organisers",
      "def": "Some results are self-reported, some are run by the organisers."
     }
    }
   },
   "real_reproducibility": {
    "label": "Across real labs",
    "values": {
     "none": {
      "label": "No evidence",
      "def": "No evidence that results hold in another lab."
     },
     "protocol-only": {
      "label": "Shared protocol only",
      "def": "Standard hardware or protocol, but no cross-site measurements."
     },
     "multi-site-measured": {
      "label": "Measured across sites",
      "def": "The same policies were evaluated at several sites and compared."
     },
     "not-applicable": {
      "label": "Not applicable",
      "def": "Scores come from simulation, so this question does not apply."
     }
    }
   },
   "skill": {
    "label": "Skill",
    "order": [
     "manipulation",
     "household",
     "navigation",
     "locomotion",
     "people",
     "humanlike",
     "safety",
     "reasoning",
     "world"
    ],
    "from_capability": {
     "manipulation": "manipulation",
     "dexterous": "manipulation",
     "bimanual": "manipulation",
     "mobile-manipulation": "household",
     "long-horizon": "household",
     "navigation": "navigation",
     "locomotion": "locomotion",
     "collaboration": "people",
     "human-likeness": "humanlike",
     "safety": "safety",
     "embodied-reasoning": "reasoning",
     "world-modeling": "world"
    },
    "values": {
     "manipulation": {
      "label": "Handling objects",
      "def": "Tasks where a robot grasps, moves, places or uses objects."
     },
     "household": {
      "label": "Household tasks",
      "def": "Tasks where a robot moves around a home and carries out chores with several steps."
     },
     "navigation": {
      "label": "Navigation",
      "def": "Tasks where a robot or agent travels to a place or an object."
     },
     "locomotion": {
      "label": "Walking and balance",
      "def": "Tasks about walking, balance and moving the whole body."
     },
     "people": {
      "label": "Working with people",
      "def": "Tasks where a robot works with humans or with other agents."
     },
     "humanlike": {
      "label": "Moving like people",
      "def": "Tests of how human-like a robot's motion looks."
     },
     "safety": {
      "label": "Safety",
      "def": "Tests of whether a system avoids harm and refuses unsafe requests."
     },
     "reasoning": {
      "label": "Reasoning",
      "def": "Questions about the physical world, answered by a model. No robot moves."
     },
     "world": {
      "label": "World models",
      "def": "Tests of how well a model predicts what happens next, usually as video."
     }
    }
   },
   "check": {
    "label": "Checked against real robots",
    "order": [
     "real",
     "checked",
     "not-checked"
    ],
    "values": {
     "real": {
      "label": "Real robots",
      "def": "The scores come from physical robots.",
      "value": "Real robots"
     },
     "checked": {
      "label": "Checked against real robots",
      "def": "The same policies were scored on this benchmark and on real robots, and the results were compared.",
      "value": "Checked"
     },
     "not-checked": {
      "label": "Not checked",
      "def": "No comparison with real robots has been published.",
      "value": "Not checked"
     }
    }
   }
  }
 },
 "benchmarks": [
  {
   "id": "1x-world-model-challenge",
   "name": "1X World Model Challenge",
   "aliases": [
    "1X World Model Compression Challenge",
    "1X World Model Sampling Challenge",
    "1xgpt"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores world models on predicting a humanoid robot's future observations from its logs and actions; built for robot policy evaluation. Matches 'world-model evaluations built for robotics'.",
   "summary": {
    "text": "Competition to predict future first-person frames of 1X's EVE robot from its logs and actions.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "1X Technologies; 2025 phases run with OpenDriveLab as part of Autonomous Grand Challenge 2025 (CVPR 2025, ICCV 2025 workshops)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-06 (GitHub repo 1xgpt created 2024-06-13; README notes v1.0 release on 2024-07-08)",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Repo creation date from GitHub API; Phase 1 blog not opened."
    },
    "latest_update": {
     "value": "ICCV 2025 phase: deadline 2025-09-27, winners presented 2025-10-19 at the Workshop on Learning to See. Home Space last modified 2025-09-19.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "OpenDriveLab is the co-organiser."
    },
    "version": {
     "value": "v2.0 data (train/val ~100 h, test v2.0 250 samples). 2025 rules switched compression to the Cosmos spatio-temporal tokenizer and sampling from 0.5 s to 2 s ahead; future actions allowed.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Predict frames/tokens from recorded logs; no control loop."
    },
    "capability": {
     "value": [
      "world-modeling"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "State vector includes hip, knee, ankle, arm, neck joints, hand closure, linear and angular velocity (raw data card)."
    },
    "scene": {
     "value": [
      "office-lab"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "~100 hours (v2.0); raw: 100 shards, 512x512 MP4 + states; test v2.0: 250 samples.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Raw data card: https://huggingface.co/datasets/1x-technologies/world_model_raw_data"
    },
    "scoring": {
     "value": [
      "fidelity",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "metric_detail": {
     "value": "Winning team's report names the compression metric 'Top-500 CE'.",
     "level": "reported",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Participant report, not organiser rules."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Official (: HF competition Spaces; public test set on the board, winners decided on a private test set. On 2026-10-10 Sampling Space RUNNING, Compression Space SLEEPING.)",
     "note": "Space status from HF API."
    },
    "top_score": {
     "value": "Revontuli: 23.0 dB PSNR (sampling), Top-500 CE 6.6386 (compression), 1st in both.",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Self-reported in arXiv 2510.07092 (v1 2025-10-08). Not comparable with the 2024 '~26.5 PSNR' guidance, which used a 0.5 s horizon."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub licence API for 1xgpt."
    },
    "license_data": {
     "value": "Raw video CC-BY-NC-SA-4.0; tokenized dataset Apache-2.0 (faces blurred in v2.0)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Card metadata; tokenized card: https://huggingface.co/datasets/1x-technologies/world_model_tokenized_data"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The planned 'Evaluation Challenge' (rank policies inside a world model) was announced as upcoming in 2024 and not found launched. Compression loss and PSNR were not linked to policy-ranking accuracy in any source found."
    },
    "citations": {
     "value": null,
     "display": "Winner report arXiv 2510.07092: 9 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Challenge itself has no paper."
    },
    "github_stars": {
     "value": 569,
     "display": "569 (1xgpt)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API; last push 2024-11-08."
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Last phase ended 2025-09-27; no 2026 phase found."
    },
    "kind": {
     "value": "challenge",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "1x-technologies/1X_World_Model_Challenge_Home on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/1x-technologies/1X_World_Model_Challenge_Home/blob/main/app.py",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s2": {
     "title": "1x-technologies/1xgpt on GitHub (repository)",
     "url": "https://github.com/1x-technologies/1xgpt",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "Challenge 2025 | OpenDriveLab",
     "url": "https://opendrivelab.com/challenge2025/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "1x-technologies/world_model_tokenized_data on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/1x-technologies/world_model_tokenized_data",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "Generative World Modelling for Humanoids: 1X World Model Challenge Technical Report",
     "url": "https://arxiv.org/abs/2510.07092",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s6": {
     "title": "1x-technologies/world_model_raw_data on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/1x-technologies/world_model_raw_data",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "1x-world-model-eval",
   "name": "1X World Model eval",
   "full_name": "1X World Model evaluation (1XWM)",
   "aliases": [
    "1X World Model: Evaluating Bits, not Atoms",
    "1XWM",
    "Redwood world model (link label on 1X challenge page)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "A learned world model used to score humanoid robot policies, built for robotics. In scope as a world-model evaluation, but closed: internal tool reported once.",
   "summary": {
    "text": "1X's internal video world model that predicts humanoid task success to rank policy checkpoints before real trials.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "1X World Model Team (contributors Daniel Ho, Jack Monas, Juntao Ren, Christina Yu), 1X Technologies",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Lead organisation's country not stated on pages read."
    },
    "first_release": {
     "value": "2025-06 (blog '1X World Model', 2025-06-16; PDF report linked from it)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Checked blog page and 1X challenge Space links."
    },
    "version": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Report carries no version or date."
    },
    "venue": {
     "value": "world-model",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Policies are run inside the learned model; e.g. Arcade run continuously for ten minutes with resets."
    },
    "capability": {
     "value": [
      "world-modeling",
      "manipulation"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Report mentions home use as goal; task settings (air fryer, object drop station, shelf) not tied to a scene type."
    },
    "scale": {
     "value": "Tasks Airfryer, Arcade, Shelf; ~216M Shelf video tokens -> alignment 63.06%; + ~1.46B Arcade tokens -> 71.17%.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "success-rate",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No code or weights linked from the blog or report."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Evaluation data not released."
    },
    "access": {
     "value": "closed",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Self-measured by 1X. Checkpoints from 2 policy training runs and several architecture variants were scored in the world model and in double-blind real A/B runs on the Arcade task; plots show agreement but no correlation statistic. Success-prediction 'alignment' 63.06% (Shelf only) -> 71.17% (with Arcade data). Analytic claim: with 70% alignment and a 15% true gap, picks the better policy 90% of the time."
    },
    "status": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10"
    },
    "kind": {
     "value": "study",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Struggles with unseen objects; pose errors accumulate in locomotion, limiting long-horizon",
     "text": "Limitations stated: struggles with unseen objects; pose errors accumulate in locomotion, limiting long-horizon navigation; without failure data, generations show optimistic bias toward success.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "https://www.1x.tech/1x-world-model.pdf",
     "url": "https://www.1x.tech/1x-world-model.pdf",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "1X World Model | 1X",
     "url": "https://www.1x.tech/discover/redwood-ai-world-model",
     "type": "blog",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "agibot-world",
   "name": "AgiBot World",
   "full_name": "AgiBot World Colosseo: A Large-scale Manipulation Platform for Scalable and Intelligent Embodied Systems",
   "aliases": [
    "AgiBot World Colosseo",
    "AgiBot World Alpha",
    "AgiBot World Beta",
    "AgiBotWorld-Alpha",
    "AgiBotWorld-Beta",
    "智元机器人 AgiBot World 百万真机数据集"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "A large real-robot training dataset. Its paper reports one-off real-robot tests run by the builders at their own site, scored with partial credit. There is no public, re-runnable evaluation or leaderboard for the dataset itself. The AgiBot World Challenge, EWMBench, Genie Sim and AGIBOT WORLD 2026 are separate records.",
   "summary": {
    "text": "AgiBot World is a real-robot training dataset from the robot maker AgiBot and partners: about 1 million sub-task trajectories, cut from about 160,000 recorded episodes, on 217 tasks, collected with about 100 identical AgiBot G1 robots in a 4,000 m² facility. Its paper scores policies on AgiBot's own robots and tasks; there is no public test of the dataset itself.",
    "sources": [
     "s2",
     "s13",
     "s4"
    ],
    "short": "AgiBot World is a training dataset recorded with about 100 real AgiBot robots. It is used to pretrain robot policies (the models that control robots), and it has no public test of its own."
   },
   "facts": {
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The paper calls AgiBot World Colosseo a 'platform' of data, models, benchmarks and ecosystem. What is released is a dataset plus a policy model (GO-1). The evaluation tasks, rubric and robots used for scoring are not released as a test others can run, so the Atlas classes it as a dataset."
    },
    "kind_secondary": {
     "value": [
      "study"
     ],
     "display": "Its scores come from in-house real-robot tests run only by the builders",
     "level": "inferred",
     "sources": [
      "s2",
      "s18",
      "s8",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "All paper evaluations ran on AgiBot G1 robots in AgiBot's facility. Asked how to evaluate without a real robot, a maintainer suggested comparing predicted actions with recorded ones (an open-loop check) and said this 'may have some discrepancies with real-world performance' (issue #42). The repo ships an open-loop evaluation script (evaluate/openloop_eval.py) and examples for LIBERO and AgileX; the README links a RoboTwin integration hosted in the RoboTwin repo. None is an AgiBot World test."
    },
    "version": {
     "value": "Beta (complete) and Alpha (subset)",
     "display": "Two releases on Hugging Face: AgiBot World Beta, the complete set, and AgiBot World Alpha, an earlier subset. No version tags in the code repository.",
     "level": "verified",
     "sources": [
      "s4",
      "s9",
      "s11",
      "s19"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AgiBot World Beta",
       "display": "Released 2025-03-01. README: 1,003,672 trajectories, about 43.8 TB. Hugging Face page: 48.1 TB total file size. Files last changed 2025-10-13.",
       "level": "verified",
       "sources": [
        "s4",
        "s9",
        "s10"
       ]
      },
      {
       "value": "AgiBot World Alpha",
       "display": "Released 2024-12-30; sample set 2025-01-03; on OpenDataLab 2025-01-20. README: 92,214 trajectories, about 8.5 TB. 36 tasks. Updated 2025-04-14 (frame-loss episodes removed, data anonymised and compressed, some camera parameters corrected). Files last changed 2025-09-29.",
       "level": "verified",
       "sources": [
        "s4",
        "s11",
        "s12",
        "s14"
       ],
       "note": "A maintainer said in 2026-05 that Alpha is a subset of Beta, collected at an earlier stage (issue #151)."
      },
      {
       "value": "Paper versions",
       "display": "arXiv v1 2025-03-09, v2 2025-03-13, v3 2025-04-30, v4 2025-08-04. v4 is the IROS 2025 camera-ready text and adds a comparison with π0.",
       "level": "verified",
       "sources": [
        "s1",
        "s2"
       ]
      },
      {
       "value": "AGIBOT WORLD 2026",
       "display": "A separate, newer dataset on AgiBot's G2 robot, released in phases from 2026-03. Separate Atlas record (agibot-world-2026).",
       "level": "verified",
       "sources": [
        "s30"
       ]
      }
     ],
     "short": "Beta (complete) and Alpha (subset)"
    },
    "publishers": {
     "value": [
      "AgiBot Inc.",
      "The University of Hong Kong",
      "Shanghai Innovation Institute",
      "Shanghai AI Lab"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s4",
      "s40"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AgiBot Inc.",
       "display": "Built the AgiBot G1 robot and the collection facility. Project co-lead Maoqing Yao (agibot.com address).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "The University of Hong Kong",
       "display": "Listed in paper v4. Not listed in v1, which named Shanghai AI Lab, AgiBot Inc. and Shanghai Innovation Institute.",
       "level": "verified",
       "sources": [
        "s2",
        "s3"
       ]
      },
      {
       "value": "Shanghai Innovation Institute",
       "display": "Project co-lead Hongyang Li (sii.edu.cn address).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Shanghai AI Lab",
       "display": "Project co-lead Yu Qiao (pjlab.org.cn address).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "OpenDriveLab (code host)",
       "display": "The code repository and project blog sit under OpenDriveLab.",
       "level": "verified",
       "sources": [
        "s4",
        "s25"
       ]
      }
     ],
     "note": "Authored as 'Team AgiBot-World', authors in alphabetical order. A Chinese news report of the 2024-12-30 launch names only AgiBot (智元机器人) as the releaser.",
     "short": "AgiBot with HKU, SII and Shanghai AI Lab"
    },
    "builder_type": {
     "value": "robot-company",
     "level": "inferred",
     "sources": [
      "s2",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "AgiBot built the hardware and the data factory and led the release; academic labs co-authored."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Shanghai and Hong Kong organisations."
    },
    "first_release": {
     "value": "2024-12",
     "display": "Alpha released 2024-12-30; complete set (Beta) 2025-03-01; paper on arXiv 2025-03-09.",
     "level": "verified",
     "sources": [
      "s4",
      "s9",
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "README news and the Hugging Face gate text both give 2024-12-30 for Alpha. The paper's change log says 'Jan 2025' for the Alpha release.",
     "short": "December 2024 (Alpha) and March 2025 (full set)"
    },
    "latest_update": {
     "value": "2025-10",
     "display": "Beta files last changed on Hugging Face 2025-10-13; Alpha 2025-09-29. Last code commit 2026-05-29 (GO-1 model loading).",
     "level": "verified",
     "sources": [
      "s10",
      "s12",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Code commits after 2025-09 concern the GO-1 model, not the data.",
     "short": "October 2025 for the data"
    },
    "published_at": {
     "value": "IROS 2025",
     "display": "IEEE/RSJ IROS 2025, pages 3549-3556, DOI 10.1109/IROS60139.2025.11247088",
     "level": "verified",
     "sources": [
      "s22",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The repo description says 'IROS 2025 Best Paper Award Finalist & IEEE TRO 2026'. Crossref has no TRO record for this paper. The TRO 2026 paper is the follow-up study 'Is Diversity All You Need for Scalable Robotic Manipulation?' (vol. 42, pp. 1872-1883, DOI 10.1109/TRO.2026.3686184), which uses AgiBot World data. The best-paper-finalist claim was not checked at an IROS source."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "dexterous",
      "long-horizon",
      "collaboration"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper: dual-arm manipulation, dexterous hands, tool use, long tasks such as 'make a coffee', multi-robot collaboration and a human-robot handover task. Episodes are typically about 30 seconds; some exceed 2 minutes."
    },
    "generalisation": {
     "value": [
      "object-pose",
      "visual",
      "language"
     ],
     "display": "In the paper's tests, each task is also run in 2 unseen setups: new object positions, visual distractors, or new wording.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper Section V-A1. The evaluation tasks themselves are drawn from the training tasks; policies are fine-tuned on task-specific demonstrations before testing.",
     "short": "New object positions, visual distractors or new wording"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper states that all evaluations were conducted in real-world scenarios."
    },
    "embodiment": {
     "value": [
      "humanoid",
      "bimanual-arm",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The AgiBot G1 has two arms, a waist and a wheeled base. Teleoperators can move the base, but the paper does not say which tasks use it, so mobile-manipulator is not tagged."
    },
    "robots": {
     "value": "AgiBot G1",
     "display": "AgiBot G1: two 7-DoF arms, mobile base, adjustable waist; gripper, 6-DoF dexterous hand or gripper with visuo-tactile sensors; 8 cameras; recorded at 30 Hz. More than 100 identical units.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Teleoperation by VR headset or whole-body motion capture.",
     "short": "About 100 AgiBot G1 robots"
    },
    "scene": {
     "value": [
      "home",
      "retail-logistics",
      "industrial",
      "office-lab",
      "mixed"
     ],
     "display": "Five domains rebuilt at full scale in one 4,000 m² facility: domestic, retail, industrial, restaurant and office",
     "level": "verified",
     "sources": [
      "s2",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "The restaurant domain has no taxonomy value, so 'mixed' is added. These are staged replicas inside AgiBot's data factory, not real homes or shops. Chinese launch report (secondary): home 40%, dining 20%, industrial 20%, retail 10%, office 10%.",
     "short": "Staged home, shop, factory, restaurant and office settings"
    },
    "tasks": {
     "value": 217,
     "display": "217 tasks, 87 skills",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper's evaluations use 6 of them (Restock Bag, Table Bussing, Pour Water, Restock Beverage, Fold Shorts, Wipe Table).",
     "short": "217 tasks"
    },
    "scenes": {
     "value": 106,
     "display": "106 scenes (paper); '100+ 1:1 replicated real-life scenarios' (README)",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "106 scenes"
    },
    "objects": {
     "value": 3000,
     "display": "Over 3,000 distinct objects",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "More than 3,000 objects"
    },
    "demonstrations": {
     "value": 1001552,
     "display": "1,001,552 trajectories and 2,976.4 hours (paper). These 'trajectories' are sub-task segments of about 160,000 recorded episodes.",
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICTS between sources; see items and issues.i1 and issues.i2.",
     "items": [
      {
       "value": "About 160,000 episodes",
       "display": "A maintainer: 'The dataset contains ~160K episodes, each divided into multiple trajectories', giving about 1M trajectories. The user counted 168,869 proprioception files but 165,745 observation videos in the 2025-03-27 Beta; the maintainer said missing videos were added on 2025-04-12.",
       "level": "verified",
       "sources": [
        "s13"
       ]
      },
      {
       "value": "1,003,672 trajectories",
       "display": "README figure for Beta (about 43.8 TB)",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "Alpha: 92,214 trajectories",
       "display": "README (about 8.5 TB). The Alpha card says '100,000+ trajectories ... 300 hours' and that the 2025-04 update raised total duration from 474.12 to 595.31 hours. A maintainer gave the episode count as 26,375 before and 34,512 after the update. The paper's experiments describe Alpha as 236 hours.",
       "level": "verified",
       "sources": [
        "s4",
        "s11",
        "s14",
        "s2"
       ]
      },
      {
       "value": "Failure-recovery data: about 1%",
       "display": "Episodes where the operator recovered from an error are kept and annotated with the cause and time.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Collection and annotation",
       "display": "Human teleoperation; local check for missing frames, then annotators verify each episode against the collection standard and add task and sub-step language annotations.",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      }
     ],
     "short": "About 1 million sub-task trajectories from about 160,000 episodes"
    },
    "scoring": {
     "value": [
      "progress"
     ],
     "display": "Normalized completion score with partial credit (the paper and AgiBot's site also call it a success rate)",
     "level": "verified",
     "sources": [
      "s2",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "Paper: 'Each episode scores 1.0 for full success, with fractional scores for partial success.' See issues.i5."
    },
    "metric_detail": {
     "value": "normalized completion score",
     "display": "Each rollout scores 1.0 for full success and a fraction for partial success; scores are averaged over 10 rollouts per task, setup and method. The partial-credit rubric for each task is not published.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Values below are read from Figures 5 to 7 of paper v4 (image figures).",
     "items": [
      {
       "value": "Pretraining data comparison",
       "display": "RDT pretrained on Open X-Embodiment vs AgiBot World Alpha vs Beta, then fine-tuned, on 3 tasks: average 0.47 / 0.68 / 0.77 in seen setups and 0.38 / 0.56 / 0.67 in unseen setups. On Table Bussing (seen) Alpha scored 0.65 and Beta 0.60.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "GO-1 comparison",
       "display": "Average over 5 tasks, 30 trials each (10 seen, 20 varied): RDT-1B 0.46, π0 0.58, GO-1 without latent planner 0.66, GO-1 0.78. Per task, GO-1 ranges from 0.60 (Restock Beverage) to 1.00 (Table Bussing).",
       "level": "verified",
       "sources": [
        "s2",
        "s27"
       ]
      },
      {
       "value": "Scaling fit",
       "display": "Out-of-the-box GO-1 performance on 4 seen tasks after pretraining on about 9.2k, 92k and 1M trajectories; power-law fit with Pearson r = 0.97 over these 3 points.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Data-quality ablation",
       "display": "RDT fine-tuned on Wipe Table: 482 unverified plus 528 verified trajectories ('All') 0.41 vs verified only 0.59.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "Partial-credit score, averaged over 10 runs per setup"
    },
    "trials": {
     "value": "10 rollouts per task, setup and method; 30 per task for GO-1",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GO-1 tests: 10 trials in a seen setup and 20 under variations or distractors per task.",
     "short": "10 per setup, and 30 per task for GO-1"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Figures 5 to 7 show single bars without error bars or confidence intervals. The team's GO-1 blog (2025-09) says real-robot test noise can exceed real improvements and describes fixing object positions and lighting to reduce it."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "All published scores come from the builders' own tests. No outside group has run the paper's evaluation tasks."
    },
    "leaderboard": {
     "value": "paper-only",
     "display": "Papers only for the dataset. The AgiBot World Challenge (2025, 2026) has its own leaderboards (separate records).",
     "level": "inferred",
     "sources": [
      "s4",
      "s9",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the README, Hugging Face cards or project page."
    },
    "license_code": {
     "value": "CC-BY-NC-SA-4.0",
     "display": "README: 'All the data and code within this repo are under CC BY-NC-SA 4.0'. The repo has no LICENSE file.",
     "level": "verified",
     "sources": [
      "s4",
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "pyproject.toml declares license = {file = 'LICENSE'}, but no such file exists; the GitHub licence endpoint returns 404."
    },
    "license_data": {
     "value": "CC-BY-NC-SA-4.0",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Paper, dataset cards and the Hugging Face gate text ('AgiBot World COMMUNITY LICENSE AGREEMENT') all name CC BY-NC-SA 4.0."
    },
    "access": {
     "value": "registration",
     "display": "Hugging Face gate with automatic approval: name, email, country, affiliation, phone, job title and research interest, plus acceptance of the licence and the AgiBot Privacy Policy. Also on OpenDataLab.",
     "level": "verified",
     "sources": [
      "s9",
      "s10",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Hub API: gated = 'auto'. The gate also records IP location.",
     "short": "Free after filling in a form with contact details"
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s4",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Reading of CC BY-NC-SA 4.0, which covers both data and code. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "display": "Scores come from real robots. AgiBot's later simulator Genie Sim is a separate record.",
     "level": "inferred",
     "sources": [
      "s2",
      "s39"
     ],
     "checked": "2026-10-10",
     "note": "The paper lists the lack of a simulation environment as a limitation and says one was in development 'to reflect real-world policy deployment outcome'. AgiBot later released Genie Sim; its paper reports R² = 0.931 between simulated and real scores for 16 π0.5 configurations on 4 tasks (Select Color, Recognize Size, Grasp Targets, Organize Items). Those are not the AgiBot World evaluation tasks, so that study is recorded under Genie Sim. No comparison of AgiBot World's own evaluation tasks in simulation and on real robots was found.",
     "short": "Scores come from real robots."
    },
    "real_reproducibility": {
     "value": "none",
     "level": "inferred",
     "sources": [
      "s2",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Evaluation tasks are set up in AgiBot's facility on AgiBot robots. The rubric is not published and no other site has reported running them."
    },
    "challenges": {
     "value": [
      "AgiBot World Challenge 2025 (IROS)",
      "AgiBot World Challenge 2026 (ICRA)"
     ],
     "display": "AgiBot's challenges supply AgiBot World data for training; scoring runs in Genie Sim, on real robots on site, or against held-out video",
     "level": "verified",
     "sources": [
      "s28",
      "s29"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AgiBot World Challenge 2025",
       "display": "The site says the challenge is 'Building on the AgiBot World benchmark'. Manipulation track: online simulation phase with the Genie Sim Benchmark, then on-site real-robot tests. World Model track: 30,000+ real-robot episodes from AgiBot World for training.",
       "level": "verified",
       "sources": [
        "s28"
       ],
       "note": "Separate Atlas records: agibot-world-challenge-r2a and agibot-world-challenge-wm."
      },
      {
       "value": "AgiBot World Challenge 2026",
       "display": "Same two-track design; the World Model track trains on the AGIBOT WORLD open dataset (30,000+ real-robot episodes); the manipulation track uses Genie Sim 3.0 online, then real robots.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      }
     ],
     "short": "AgiBot World Challenge 2025 and 2026"
    },
    "derived_benchmarks": {
     "value": [
      "EWMBench",
      "MMSI-Bench",
      "Cosmos-Reason1 embodied reasoning set"
     ],
     "display": "Benchmarks that use AgiBot World recordings as test material",
     "level": "verified",
     "sources": [
      "s36",
      "s37",
      "s38"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "EWMBench",
       "display": "2025-05, AgiBot and others. Scores robot video generators against 10 tasks of AgiBot World episodes.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "MMSI-Bench",
       "display": "2025-05. Multi-image spatial questions; AgiBot-World is one of eight image sources.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      },
      {
       "value": "Cosmos-Reason1 embodied reasoning set",
       "display": "NVIDIA, 2025. Next-subtask questions built from AgiBot World (36 tasks).",
       "level": "verified",
       "sources": [
        "s38"
       ]
      }
     ],
     "short": "3 benchmarks reuse its recordings"
    },
    "citations": {
     "value": 509,
     "display": "509 (Semantic Scholar; 65 influential)",
     "level": "verified",
     "sources": [
      "s24"
     ],
     "checked": "2026-10-10",
     "short": "509"
    },
    "github_stars": {
     "value": 3204,
     "display": "3,204 stars, 220 forks (OpenDriveLab/AgiBot-World)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "3,204"
    },
    "dataset_downloads": {
     "value": 76493,
     "display": "Beta: 76,493 (Hub 'downloads' field), 1,113,897 all time, 83 likes. Alpha: 29,151, 286,342 all time, 239 likes.",
     "level": "verified",
     "sources": [
      "s10",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hugging Face Hub API. OpenDataLab downloads not counted.",
     "short": "76,493 for Beta on Hugging Face"
    },
    "used_by": {
     "value": "Used as pretraining data by AgiBot's GO-1 and Genie Envisioner, NVIDIA's GR00T N1, villa-X and EO-1, among others.",
     "level": "verified",
     "sources": [
      "s2",
      "s31",
      "s32",
      "s33",
      "s34",
      "s35"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list. These reports use the data for training; none reports scores on an AgiBot World test.",
     "items": [
      {
       "value": "GO-1",
       "display": "AgiBot, 2025-03 paper; model open-sourced 2025-09-19.",
       "level": "verified",
       "sources": [
        "s2",
        "s4"
       ]
      },
      {
       "value": "GR00T N1",
       "display": "NVIDIA, 2025-03. Pretraining used 140,000 AgiBot-Alpha trajectories 'available at the time'.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "Is Diversity All You Need for Scalable Robotic Manipulation? (GO-1-Pro)",
       "display": "AgiBot and OpenDriveLab, 2025-07; IEEE TRO 2026. Builds its pretraining sets from AgiBot World Beta.",
       "level": "verified",
       "sources": [
        "s34",
        "s23"
       ]
      },
      {
       "value": "villa-X",
       "display": "2025-07. Pretraining mixture includes AgiBot World Beta.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "Genie Envisioner",
       "display": "AgiBot, 2025-08. World model pretrained on AgiBot-World-Beta.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "EO-1",
       "display": "2025-08. Trained on AgiBotWorld, Open X-Embodiment, RoboMIND and other data.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      }
     ],
     "short": "Pretraining data for GO-1, GR00T N1 and others"
    },
    "industry_use": {
     "value": [
      "AgiBot",
      "NVIDIA"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s32",
      "s31",
      "s38"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AgiBot",
       "display": "Trains GO-1 and Genie Envisioner on it and builds its challenges on it.",
       "level": "verified",
       "sources": [
        "s2",
        "s32",
        "s28"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "GR00T N1 pretraining data; source for Cosmos-Reason1 embodied reasoning questions.",
       "level": "verified",
       "sources": [
        "s31",
        "s38"
       ]
      }
     ]
    },
    "status": {
     "value": "maintained",
     "display": "Data last changed 2025-10; code for the GO-1 model updated until 2026-05. AgiBot's new data goes into the separate AGIBOT WORLD 2026 release.",
     "level": "inferred",
     "sources": [
      "s10",
      "s12",
      "s6",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "38 open issues on GitHub (API count, includes pull requests).",
     "short": "Maintained. New data goes into the separate AGIBOT WORLD 2026 release."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "The figure of 1 million trajectories counts sub-task segments",
     "text": "A maintainer explained that the dataset holds about 160,000 episodes, each split into several sub-task 'trajectories' (action slices), which gives the 1 million figure. The paper's comparison table sets this 1M+ against counts for other datasets such as DROID (76k) and RoboMIND (55k) without saying the unit differs. Counted as recorded episodes, AgiBot World is about 160,000.",
     "level": "verified",
     "sources": [
      "s13",
      "s2"
     ],
     "status": "open",
     "note": "Whether the other datasets in Table I count whole episodes was not checked for each one.",
     "short": "The figure of 1 million trajectories counts sub-task pieces cut from about 160,000 recorded episodes."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "Size figures differ between sources",
     "text": "Beta: 1,001,552 trajectories and 2,976.4 hours in the paper; 1,003,672 trajectories in the README; 'approximately one million ... episodes, totaling 2,967 hours' in AgiBot's Genie Envisioner paper. Alpha: 92,214 trajectories (README); '100,000+ trajectories ... 300 hours' (card); 236 hours (paper experiments); 474.12 hours rising to 595.31 hours after the 2025-04 update (card); 26,375 rising to 34,512 episodes (maintainer). GR00T N1 used '140,000 trajectories' of Alpha. The paper also calls Alpha 'roughly 14%' of Beta's trajectories in the text and 'around 10%' in its change log; 92,214 is 9.2% of 1,001,552.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s11",
      "s14",
      "s31",
      "s32"
     ],
     "status": "open",
     "short": "The trajectory and hour counts for both releases differ between the paper, the README, the dataset cards and later papers."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "The camera and action fields have known problems",
     "text": "In 2025-03 a maintainer confirmed errors in the camera extrinsic parameters and promised corrected data; the 2025-04-14 Alpha update corrected 'some data'. Users still reported misaligned projections in 2025-11 (issue #122) and 2026-02 (issue #28, open). The end-effector pose fields under 'action' equal those under 'state' in whole episodes; a maintainer said some VR controller poses were not recorded and were copied from the state, recommended joint representations, and said GO-1 is trained to predict the state as the action. A user reported frame and video misalignment in 9 tasks after conversion to LeRobot 3.0 (issue #149, open, no maintainer reply).",
     "level": "verified",
     "sources": [
      "s15",
      "s16",
      "s17",
      "s14",
      "s41"
     ],
     "status": "open",
     "note": "The 9-task misalignment is a user report only.",
     "short": "The maintainers have acknowledged errors in the camera calibration. They also confirmed that some action fields were copied from the robot state. The fixes so far are partial."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The evaluation cannot be re-run by others",
     "text": "The paper's tests use 6 tasks staged in AgiBot's facility, AgiBot G1 robots, task-specific fine-tuning data and an unpublished partial-credit rubric. No simulator, leaderboard or test server exists for the dataset. The Hugging Face card still lists 'AgiBot World Colosseum: Comprehensive platform (expected release date: 2025)' as a to-do. For offline checks a maintainer suggested comparing predicted with recorded actions, adding that this may differ from real-world performance. AgiBot's later evaluation work moved to Genie Sim and the AgiBot World Challenge.",
     "level": "inferred",
     "sources": [
      "s2",
      "s9",
      "s18",
      "s28"
     ],
     "status": "open",
     "note": "Inferred from the paper, cards, README and issues; no statement by the authors that the tests are closed.",
     "short": "The paper's tests ran at AgiBot's site and used a scoring rubric that is not published. No one else can repeat them."
    },
    {
     "id": "i5",
     "type": "inconsistent-reporting",
     "title": "Completion scores are presented as success rates",
     "text": "Figure 5 reports normalized completion scores with partial credit (average RDT-1B 0.46, GO-1 0.78). AgiBot's GO-1 page describes the same numbers as 'increasing success rates by 32% (46% → 78%)', and the abstract says GO-1 achieves 'over 60% success rate on complex tasks'. The abstract's '30%' improvement over Open X-Embodiment is an absolute gain of 0.30 in completion score (0.47 to 0.77), not a relative gain.",
     "level": "verified",
     "sources": [
      "s2",
      "s27"
     ],
     "status": "open",
     "short": "The paper's abstract and AgiBot's website describe scores that give partial credit as success rates."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Early data showed the faces of robot operators",
     "text": "In 2025-01 a user noted that faces were not blurred in a rear camera of the sample data. AgiBot replied that it had an agreement with the teleoperators and had attempted blurring. The 2025-04-14 Alpha update says the data was anonymised to remove personal and sensitive information. Task 429 (memory-kit insertion) was removed from Alpha in 2025-01 because of a visuo-tactile sensor setup problem, to be re-collected.",
     "level": "verified",
     "sources": [
      "s20",
      "s21",
      "s11"
     ],
     "status": "addressed",
     "short": "Early data showed the faces of the people who operated the robots remotely. The April 2025 update anonymised the data."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "AgiBot World is training data. It is not a benchmark. Its paper's scores show that pretraining on it (training on it before training for a specific task) helped AgiBot's policies on AgiBot's own tasks and robots. Nobody else can run those tests, so the scores cannot be compared with other papers.",
     "basis": [
      "facts.kind",
      "facts.evaluator",
      "facts.real_reproducibility",
      "issues.i4"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "AgiBot World is training data, and its tests ran only inside AgiBot. Other teams cannot run them as a benchmark."
    },
    {
     "id": "r2",
     "text": "The comparison with Open X-Embodiment (another large robot dataset) was run on the same robot and in the same facility that produced AgiBot World's data. A gain there may come from the quality of the data or from testing on the robot and site that recorded it. An outside test on other robots would be needed to separate the two.",
     "basis": [
      "facts.metric_detail",
      "facts.robots",
      "issues.i4"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The test that compares AgiBot World with other data favours AgiBot's own robot and site."
    },
    {
     "id": "r3",
     "text": "When you quote its size, give the unit. It holds about 160,000 episodes, or about 1 million sub-task trajectories. The headline counts and hours differ between the paper, the README, the dataset cards and later papers.",
     "basis": [
      "facts.demonstrations",
      "issues.i1",
      "issues.i2"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "When you quote its size, say whether you mean episodes or sub-task trajectories."
    },
    {
     "id": "r4",
     "text": "The scaling result (Pearson r = 0.97 for a power-law fit) is based on three pretraining sizes. It shows that more data helped in these tests. It is weak evidence for a general scaling law (a fixed rule linking the amount of data to performance).",
     "basis": [
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The claim about how results grow with more data is based on three data points."
    }
   ],
   "searched": [
    {
     "for": "validity (sim-to-real or cross-site comparisons of AgiBot World scores)",
     "where": "Paper v1 and v4 full text and figures; README; Hugging Face cards; OpenDriveLab project page and GO-1 blog; agibot-world.com page text (JS bundles for the dataset, GO-1, Genie Sim and both challenges); Genie Sim 3.0 paper v4 (its sim-real study uses other tasks); GitHub issues (160 issues and pull requests, titles read, key threads opened). No paired comparison involving AgiBot World's own evaluation tasks was found. Scores already come from real robots.",
     "date": "2026-10-10"
    },
    {
     "for": "evaluation rubric and leaderboard",
     "where": "Paper Section V-A, README, both Hugging Face cards, project page, agibot-world.com bundles, GitHub evaluate/ folder (deploy.py, openloop_eval.py, LIBERO and AgileX examples). No per-task partial-credit rubric and no leaderboard for the dataset.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code",
     "where": "GitHub contents listing (no LICENSE file), GitHub licence API (404), pyproject.toml, README licence section.",
     "date": "2026-10-10"
    },
    {
     "for": "published_at (IEEE TRO 2026 and best-paper-finalist claims)",
     "where": "Crossref search for the paper title (only the IROS 2025 record) and for the follow-up title (TRO 2026 record); Semantic Scholar record of the follow-up (venue IEEE Transactions on Robotics). The IROS award claim was not checked: the shared web-search budget was used up before an IROS source could be found.",
     "date": "2026-10-10"
    },
    {
     "for": "official Chinese launch announcement",
     "where": "agibot-world.com bundles (English and Chinese strings). A Chinese news report (IT之家, 2024-12-30) was read; thepaper.cn returned HTTP 403. No AgiBot press page was opened.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "AgiBot World Colosseo (arXiv abstract page, submission history v1 to v4)",
     "url": "https://arxiv.org/abs/2503.06669",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "AgiBot World Colosseo, full text v4 (incl. Figures 5 to 7)",
     "url": "https://arxiv.org/html/2503.06669v4",
     "type": "paper",
     "publisher": "arXiv (IROS 2025 camera-ready text)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "AgiBot World Colosseo, full text v1",
     "url": "https://arxiv.org/html/2503.06669v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "OpenDriveLab/AgiBot-World README",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/blob/main/README.md",
     "type": "repo",
     "publisher": "OpenDriveLab / AgiBot",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "GitHub API: OpenDriveLab/AgiBot-World (stars, forks, licence)",
     "url": "https://api.github.com/repos/OpenDriveLab/AgiBot-World",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "OpenDriveLab/AgiBot-World commit history",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/commits/main",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2026-05-29",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "AgiBot-World pyproject.toml (license = file LICENSE)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/blob/main/pyproject.toml",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "AgiBot-World evaluate/ folder (deploy.py, openloop_eval.py, LIBERO and AgileX examples)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/tree/main/evaluate",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "agibot-world/AgiBotWorld-Beta dataset card and gate",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorld-Beta",
     "type": "repo",
     "publisher": "AgiBot World",
     "date": "2025-10-13",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "Hugging Face Hub API: agibot-world/AgiBotWorld-Beta (downloads, likes, gated, lastModified)",
     "url": "https://huggingface.co/api/datasets/agibot-world/AgiBotWorld-Beta?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated&expand[]=lastModified&expand[]=cardData",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "agibot-world/AgiBotWorld-Alpha dataset card (incl. 'Important Notice' update log)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorld-Alpha",
     "type": "repo",
     "publisher": "AgiBot World",
     "date": "2025-09-29",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Hugging Face Hub API: agibot-world/AgiBotWorld-Alpha",
     "url": "https://huggingface.co/api/datasets/agibot-world/AgiBotWorld-Alpha?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated&expand[]=lastModified",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "AgiBot-World issue #79: 'Dataset Incomplete? Only 160k Trajectories Found Instead of 1M' (maintainer reply)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/79",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "AgiBot-World issue #43: 'Change log requirement' (Alpha update 2025-04-14)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/43",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "AgiBot-World issue #28: 'Error in camera extrinsic' (open)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/28",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "AgiBot-World issue #122: inaccurate extrinsic parameters (open)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/122",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "AgiBot-World issue #120: 'state and action are entirely the same' (maintainer reply)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/120",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "AgiBot-World issue #42: 'How to evaluate the performance without deploying on a real robot?' (maintainer reply)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/42",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "AgiBot-World issue #151: Alpha a subset of Beta? (maintainer reply)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/151",
     "type": "repo",
     "publisher": "OpenDriveLab",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "AgiBotWorld-Alpha discussion #23: removing personally identifiable information (AgiBot reply)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorld-Alpha/discussions/23",
     "type": "repo",
     "publisher": "AgiBot World",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "AgiBotWorld-Alpha discussion #18: task 429 removed (AgiBot reply)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorld-Alpha/discussions/18",
     "type": "repo",
     "publisher": "AgiBot World",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Crossref record: AgiBot World Colosseo, IROS 2025, DOI 10.1109/IROS60139.2025.11247088",
     "url": "https://api.crossref.org/works?query.bibliographic=AgiBot+World+Colosseo+large-scale+manipulation+platform+scalable+intelligent+embodied+systems&rows=6",
     "type": "index",
     "publisher": "Crossref",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Crossref record: 'Is Diversity All You Need for Scalable Robotic Manipulation?', IEEE TRO vol. 42, DOI 10.1109/TRO.2026.3686184",
     "url": "https://api.crossref.org/works?query.bibliographic=Is+Diversity+All+You+Need+for+Scalable+Robotic+Manipulation&rows=4",
     "type": "index",
     "publisher": "Crossref",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Semantic Scholar API record for arXiv:2503.06669",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2503.06669?fields=title,citationCount,influentialCitationCount,venue,publicationDate,externalIds",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "AgiBot World Colosseo project page",
     "url": "https://opendrivelab.com/AgiBot-World/",
     "type": "site",
     "publisher": "OpenDriveLab",
     "date": "2025-03-10",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Open-sourcing GO-1: The Bitter Lessons of Building VLA Systems at Scale (blog)",
     "url": "https://opendrivelab.com/OpenGO1/",
     "type": "site",
     "publisher": "OpenDriveLab",
     "date": "2025-09-19",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "agibot-world.com GO-1 page text (site bundle)",
     "url": "https://agibot-world.com/assets/index-Btm4Rpca.js",
     "type": "site",
     "publisher": "AgiBot",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "agibot-world.com AgiBot World Challenge 2025 page text (site bundle)",
     "url": "https://agibot-world.com/assets/index-CKedb3O3.js",
     "type": "site",
     "publisher": "AgiBot",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "agibot-world.com AgiBot World Challenge 2026 page text (site bundle)",
     "url": "https://agibot-world.com/assets/index-D6a26r2O.js",
     "type": "site",
     "publisher": "AgiBot",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "agibot-world.com main site bundle (dataset, AGIBOT WORLD 2026 and Genie Sim text)",
     "url": "https://agibot-world.com/assets/index-B6M2W_mF.js",
     "type": "site",
     "publisher": "AgiBot",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots (pretraining data section)",
     "url": "https://arxiv.org/html/2503.14734",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
     "url": "https://arxiv.org/html/2508.05635",
     "type": "paper",
     "publisher": "arXiv (AgiBot Genie Team)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "EO-1: An Open Unified Embodied Foundation Model for General Robot Control (training data table)",
     "url": "https://arxiv.org/html/2508.21112",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Is Diversity All You Need for Scalable Robotic Manipulation?",
     "url": "https://arxiv.org/html/2507.06219",
     "type": "paper",
     "publisher": "arXiv (AgiBot, OpenDriveLab); IEEE TRO 2026",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "villa-X: Enhancing Latent Action Modeling in Vision-Language-Action Models (pretraining data)",
     "url": "https://arxiv.org/html/2507.23682",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models",
     "url": "https://arxiv.org/html/2505.09694v2",
     "type": "paper",
     "publisher": "arXiv (AgiBot, SJTU, MMLab-CUHK, HIT)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "MMSI-Bench: A Benchmark for Multi-Image Spatial Intelligence (data sources)",
     "url": "https://arxiv.org/html/2505.23764",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "Cosmos-Reason1: From Physical Common Sense To Embodied Reasoning (AgiBot data section)",
     "url": "https://arxiv.org/html/2503.15558v3",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "Genie Sim 3.0 paper v4 (GenieSim-Sim2Real section, Fig. 7)",
     "url": "https://arxiv.org/html/2601.02078v4",
     "type": "paper",
     "publisher": "arXiv (AgiBot)",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "智元机器人开源百万真机数据集 AgiBot World (news report of the launch)",
     "url": "https://www.ithome.com/0/821/109.htm",
     "type": "secondary",
     "publisher": "IT之家",
     "date": "2024-12-30",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "AgiBot-World issue #149: frame and video misalignment in 9 tasks after LeRobot 3.0 conversion (user report, open)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/issues/149",
     "type": "repo",
     "publisher": "OpenDriveLab (user report)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Expanded to a full entry from primary sources: paper v1 and v4 incl. figures, README, Hugging Face cards and API, GitHub issues, AgiBot site text, Crossref, Semantic Scholar and papers that reuse the data. Added the episode-versus-trajectory definition, size conflicts, data-quality issues, the partial-credit metric, paper results, challenges and reuse. Corrected: 'IEEE TRO 2026' belongs to the follow-up paper; scoring set to progress only."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How the tested policies would score on anyone else's robots.",
     "sub": "All of the tests ran on AgiBot robots at AgiBot's site.",
     "basis": [
      "facts.real_reproducibility",
      "issues.i4"
     ]
    },
    {
     "id": "l2",
     "text": "How a result compares with public results from other teams.",
     "sub": "There is no public test, no leaderboard and no published scoring rubric.",
     "basis": [
      "facts.leaderboard",
      "facts.evaluator",
      "issues.i4"
     ]
    },
    {
     "id": "l3",
     "text": "How good the data is for training policies on other hardware.",
     "sub": "The gains from training on it were measured on the same robot that recorded the data.",
     "basis": [
      "facts.metric_detail",
      "facts.robots"
     ]
    }
   ],
   "validity": []
  },
  {
   "id": "agibot-world-2026",
   "name": "AgiBot World 2026",
   "aliases": [
    "AGIBOT WORLD 2026",
    "AgiBotWorld2026"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "New real-robot dataset edition on different hardware (AGIBOT G2). We found no paper, no evaluation protocol and no use as an evaluation reference. It is listed so readers can tell it apart from the 2024-25 AgiBot World dataset.",
   "summary": {
    "text": "AgiBot's 2026 real-robot dataset on its G2 robot, released in themed phases; no evaluation protocol.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Released by the agibot-world Hugging Face org; the card names no authors or institutions. The site contact is an agibot.com address.",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Site text read from its JS bundle https://agibot-world.com/assets/index-B6M2W_mF.js"
    },
    "first_release": {
     "value": "2026-03: HF repo created 2026-03-11; example data path dated 20260315",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-09-01: README_THEME_3 uploaded; ReinforcementLearning rollouts uploaded 2026-08-31",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "HF tree has ImitationLearning, RichInteraction, ReinforcementLearning and simulation folders. The site says the dataset 'will be released in five sequential phases'.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Phase statement from the site bundle: https://agibot-world.com/assets/index-B6M2W_mF.js"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "long-horizon",
      "navigation",
      "collaboration"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Read from the English strings in the site's JS bundle (the site is client-rendered)."
    },
    "embodiment": {
     "value": [
      "bimanual-arm",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Mobility of the base not confirmed in text we read, so mobile-manipulator is not tagged."
    },
    "scene": {
     "value": [
      "home",
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked at the HF card, HF API metadata and site bundle. HF size tag '1K<n<10K' is metadata, not a stated count."
    },
    "scoring": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked at the card and site."
    },
    "leaderboard": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "None (found)"
    },
    "license_code": {
     "value": "CC-BY-NC-SA-4.0 (card: all data and code in this repo)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "CC-BY-NC-SA-4.0",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "access": {
     "value": null,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "open: the HF dataset is not gated"
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Real robots (no scores)",
     "note": "Real-world data with no scoring. The card says digital-twin simulation data was collected and released in the GenieSim project; no sim/real evaluation link was found."
    },
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "agibot-world/AgiBotWorld2026 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorld2026/raw/main/README.md",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s2": {
     "title": "agibot-world/AgiBotWorld2026 on Hugging Face (dataset)",
     "url": "https://huggingface.co/api/datasets/agibot-world/AgiBotWorld2026",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s3": {
     "title": "agibot-world/AgiBotWorld2026 on Hugging Face (dataset)",
     "url": "https://huggingface.co/api/datasets/agibot-world/AgiBotWorld2026/commits/main",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s4": {
     "title": "agibot-world/AgiBotWorld2026 on Hugging Face (dataset)",
     "url": "https://huggingface.co/api/datasets/agibot-world/AgiBotWorld2026/tree/main",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "https://agibot-world.com/assets/index-B6M2W_mF.js",
     "url": "https://agibot-world.com/assets/index-B6M2W_mF.js",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "https://agibot-world.com/",
     "url": "https://agibot-world.com/",
     "type": "paper",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "agibot-world-challenge-r2a",
   "name": "AgiBot World Challenge · R2A",
   "full_name": "AgiBot World Challenge — Manipulation / Reasoning to Action track",
   "aliases": [
    "AgiBot World Challenge @ IROS 2025 Manipulation track",
    "AGIBOT WORLD CHALLENGE 2026 @ ICRA Reasoning to Action (R2A)",
    "Reasoning2Action"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Time-bound competition that scores manipulation policies, online in simulation and onsite on real robots.",
   "summary": {
    "text": "AgiBot manipulation contest (2025, 2026): online simulation ranking on 10 tasks, then finalists run on real robots.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2025 co-hosted by AgiBot and OpenDriveLab; organisers listed from AgiBot, HKU and OpenDriveLab.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "2026 hosted by AgiBot; recap does not name OpenDriveLab.",
       "level": "verified",
       "sources": [
        "s3"
       ]
      }
     ]
    },
    "first_release": {
     "value": "2025-05 (challenge start and tutorial 'late May'; HF dataset repo created 2025-05-26; dataset v1.0.0 2025-06-25).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-06-05 AgiBot recap of the ICRA 2026 final; plans an online simulation leaderboard and more tasks.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim+real",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid",
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Mobile-chassis detail from https://agibot-world.com/ (AGIBOT WORLD 2026 dataset text). 2025 prize vouchers could buy 'one AgiBot G1 or G2 robot'."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home",
      "kitchen",
      "retail-logistics",
      "industrial",
      "office-lab"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2025 tasks (10): heat food in microwave, open drawer and store items, pack in supermarket, stamp the seal, pack moving objects from conveyor, clear countertop waste, pick up items from freezer, restock supermarket items, make a sandwich, clear restaurant table; sim data + real-robot data (hundreds of trajectories per task).",
       "level": "verified",
       "sources": [
        "s5"
       ]
      },
      {
       "value": "2026 tasks (10): clean_the_desktop, hold_pot, open_door, place_block_into_box, pour_workpiece, scoop_popcorn, sorting_packages, sorting_packages_continuous, stock_and_straighten_shelf, take_wrong_item_shelf.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "Participation: 2025 - 431 teams, 23 countries/regions (both tracks). 2026 - 526 teams, 27 countries; >100 teams beat the baseline.",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "note": "2025 figure from AgiBot press release https://www.prnewswire.com/news-releases/agibot-robotics-debuted-at-iros-2025-the-agibot-world-challenge-concluded-successfully-302599546.html"
      },
      {
       "value": "Finalists 2025: official page says top ten teams; press release says 11 teams advanced.",
       "level": "verified",
       "sources": [
        "s4"
       ],
       "note": "Conflict between two AgiBot sources."
      }
     ]
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "display": "Official (2026 public online leaderboard with per-task scores, 79 teams; top: GreenVLA 8.85, SynapX 8.48, RP_VLA 8.47.)",
     "note": "2025 online leaderboard not retrievable on 2026-10-10."
    },
    "license_data": {
     "value": "Challenge datasets 2025 and 2026: CC BY-NC-SA 4.0 (HF dataset cards). Genie Sim assets: CC BY-NC-SA 4.0.",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Assets: https://huggingface.co/datasets/agibot-world/GenieSimAssets"
    },
    "license_code": {
     "value": "2026 baseline ACoT-VLA Apache-2.0; Genie Sim core code MPL-2.0 (scene_reconstruction has mixed licences).",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "ACoT-VLA licence from GitHub API: https://github.com/AgibotTech/ACoT-VLA"
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Tried, not measured (Challenge itself: finalists' policies run in sim and on real robots, no published correlation.)",
     "note": "Top online (sim) teams were re-run on real robots onsite, but no sim-vs-real statistic was published for the challenge. 2026 online #1 GreenVLA (8.85) was listed third onsite; onsite champion PrismBot ranked 10th online (7.93). Platform-level: Genie Sim 3.0 paper reports R^2=0.931 over 16 pi0.5 configurations (AgiBot-measured)."
    },
    "kind": {
     "value": "challenge",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Prismbot 10th online, greenvla 1st online but not champion. many teams score 1.0 on severa",
     "text": "Our reading of leaderboard + recap; onsite tasks differ partly from online tasks.",
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "AGIBOT WORLD",
     "url": "https://agibot-world.com/challenge2026",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Challenge 2025 | OpenDriveLab",
     "url": "https://opendrivelab.com/challenge2025/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "AGIBOT WORLD CHALLENGE 2026 Advances Embodied AI Competition from Simulation to Real-AGIBOT Innovation (Shanghai) Technology Co., Ltd.",
     "url": "https://agibot.com/article/231/detail/73.html",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "AgiBot Robotics Debuted at IROS 2025, the AgiBot World Challenge Concluded Successfully",
     "url": "https://www.prnewswire.com/news-releases/agibot-robotics-debuted-at-iros-2025-the-agibot-world-challenge-concluded-successfully-302599546.html",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "agibot-world/AgiBotWorldChallenge-2025 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorldChallenge-2025",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "agibot-world/AgiBotWorldChallenge-2026 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorldChallenge-2026",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s7": {
     "title": "Reasoning2Action | AgiBot World Challenge",
     "url": "https://agibot-world.com/challenge2026/reasoning2action/quick-start",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Reasoning2Action | AgiBot World Challenge",
     "url": "https://agibot-world.com/challenge2026/reasoning2action/leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "AgibotTech/genie_sim on GitHub (repository)",
     "url": "https://github.com/AgibotTech/genie_sim",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "Genie Sim 3.0 : A High-Fidelity Comprehensive Simulation Platform for Humanoid Robot (full text)",
     "url": "https://arxiv.org/html/2601.02078v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "agibot-world-challenge-wm",
   "name": "AgiBot World Challenge · WM",
   "full_name": "AgiBot World Challenge — World Model track",
   "aliases": [
    "AgiBot World Challenge @ IROS 2025 World Model track",
    "AGIBOT WORLD CHALLENGE 2026 @ ICRA World Model track",
    "ICRA26WM"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "World-model evaluation built for robotics: action-conditioned prediction of a robot's camera view.",
   "summary": {
    "text": "AgiBot contest track (2025, 2026): predict robot head-camera video from actions; scored against held-out real video.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "2025 co-hosted by AgiBot and OpenDriveLab; World Model contact from AgiBot. 2026 hosted by AgiBot.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-05 (baseline repo created 2025-05-14; challenge start late May 2025).",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026 submission deadline 2026-04-20; final rankings 2026-04-30; top three invited to ICRA (2026-06-01).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "kitchen",
      "tabletop",
      "home"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2025 dataset card: train 30,000+ trajectories from 10 tasks; validation 30 samples; test 30 non-public samples mixing seen/unseen scenes, expert and failed trajectories.",
       "level": "verified",
       "sources": [
        "s5"
       ]
      },
      {
       "value": "Official page: 10 manipulation tasks, over 30 action sequences with initial images.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      }
     ]
    },
    "scoring": {
     "value": [
      "fidelity",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "README at commit 62eb4ebfec (2025-07-15)."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Official (2026 HF competition leaderboard: 153 teams; top NeoVerse-ABot 0.829, PAI@IAII 0.8245, Loop 0.8241.)"
    },
    "license_code": {
     "value": "Baseline repo README: all data and code CC BY-NC-SA 4.0; GitHub detects no LICENSE file. EWMBench (metrics) CC BY-NC-SA 4.0.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "EWMBench: https://github.com/AgibotTech/EWMBench"
    },
    "license_data": {
     "value": "CC BY-NC-SA 4.0 (HF dataset cards 2025 and 2026).",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "Not checked (No study linking this track's score to downstream robot performance found.)",
     "note": "Scores compare generated video to held-out real video; no evidence found that a higher track score predicts better robot policy evaluation or control."
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "capability": {
     "value": [
      "world-modeling"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "kind": {
     "value": "challenge",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Top 14 of 153 teams within 0.020 of each other (0.829 to 0.809); test set ~30 samples; no ",
     "text": "Our reading; test-set size from 2025 card.",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "AGIBOT WORLD",
     "url": "https://agibot-world.com/challenge2026",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Challenge 2025 | OpenDriveLab",
     "url": "https://opendrivelab.com/challenge2025/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "AgibotTech/AgiBotWorldChallengeICRA2026-WorldModelBaseline on GitHub (repository)",
     "url": "https://github.com/AgibotTech/AgiBotWorldChallengeICRA2026-WorldModelBaseline",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "Challenge Description - ICRA26 Workshop",
     "url": "https://agibot-world-icra26wm.hf.space/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "agibot-world/AgiBotWorldChallenge-2025 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/agibot-world/AgiBotWorldChallenge-2025",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "AgibotTech/AgiBotWorldChallengeICRA2026-WorldModelBaseline on GitHub (file README.md)",
     "url": "https://github.com/AgibotTech/AgiBotWorldChallengeICRA2026-WorldModelBaseline/blob/62eb4ebfec/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "Leaderboard - ICRA26 Workshop",
     "url": "https://agibot-world-icra26wm.hf.space/leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "AgiBot Robotics Debuted at IROS 2025, the AgiBot World Challenge Concluded Successfully",
     "url": "https://www.prnewswire.com/news-releases/agibot-robotics-debuted-at-iros-2025-the-agibot-world-challenge-concluded-successfully-302599546.html",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "AGIBOT WORLD CHALLENGE 2026 Advances Embodied AI Competition from Simulation to Real-AGIBOT Innovation (Shanghai) Technology Co., Ltd.",
     "url": "https://agibot.com/article/231/detail/73.html",
     "type": "blog",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "ai2-thor",
   "name": "AI2-THOR",
   "aliases": [
    "THOR",
    "iTHOR"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Simulator that hosts several scored embodied benchmarks (RoboTHOR ObjectNav, Rearrangement, ArmPointNav, CHORES); fits kind 'platform'.",
   "summary": {
    "text": "Unity-based simulator of interactive indoor rooms and houses used to train and test navigation and manipulation agents.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Allen Institute for AI (repo in the allenai GitHub organisation; 13 authors incl. Kolve, Mottaghi, Kembhavi, Farhadi)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Non-profit research institute; closest id is academic. Checked 2026-10-10."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2017-12 (arXiv v1 2017-12-14; repo created 2017-10-17)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "GitHub release 5.0.0 on 2022-12-13 (PyPI latest 5.0.0, same day); arXiv v4 2022-08-26; latest default-branch commit 2025-05-29 ('Update to ovens')",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "PyPI: https://pypi.org/project/ai2thor/ Checked 2026-10-10."
    },
    "version": {
     "value": "5.0.0",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "arXiv only (Semantic Scholar venue 'arXiv.org')",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "navigation",
      "manipulation",
      "mobile-manipulation",
      "collaboration"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Site: navigation and object interaction, robotic arm (ManipulaTHOR), multiple agents in one scene. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-base",
      "mobile-manipulator",
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LoCoBot with Kinect camera (RoboTHOR); humanoid and drone agents in iTHOR; arm agent in ManipulaTHOR; Stretch RE-1 in SPOC/CHORES. Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "kitchen",
      "home"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scale": {
     "value": "iTHOR 120 rooms with 2,000+ unique objects; RoboTHOR 89 apartments with 600+ objects",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scoring": {
     "value": [],
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "None (of its own; challenges built on it used leaderboard.allenai.org, which did not resolve (DNS) on 2026-10-10)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. Checked 2026-10-10."
    },
    "citations": {
     "value": 1585,
     "display": "1585",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 1810,
     "display": "1810",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Maintained (low activity): no release since 2022-12; commits through 2025-05)",
     "note": "Checked 2026-10-10."
    },
    "license_data": {
     "value": "No dataset of its own; scenes and assets ship in the Apache-2.0 repository/builds, no separate licence found",
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (at platform level; see RoboTHOR ObjectNav and CHORES records for benchmark-level evidence)",
     "note": "Platform level. Sim-to-real evidence exists only for specific benchmarks on it: RoboTHOR (1 model on LoCoBot) and CHORES/SPOC (2 models on Stretch), recorded separately."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "AI2-THOR: An Interactive 3D Environment for Visual AI",
     "url": "https://arxiv.org/abs/1712.05474",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2017-12"
    },
    "s2": {
     "title": "allenai/ai2thor on GitHub (repository)",
     "url": "https://github.com/allenai/ai2thor",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "allenai/ai2thor on GitHub (releases)",
     "url": "https://github.com/allenai/ai2thor/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "Client Challenge",
     "url": "https://pypi.org/project/ai2thor/",
     "type": "repo",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "ProcTHOR: Large-Scale Embodied AI Using Procedural Generation (full text)",
     "url": "https://arxiv.org/html/2206.06994",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-06"
    },
    "s6": {
     "title": "https://ai2thor.allenai.org/",
     "url": "https://ai2thor.allenai.org/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "https://ai2thor.allenai.org/robothor/cvpr-2021-challenge/",
     "url": "https://ai2thor.allenai.org/robothor/cvpr-2021-challenge/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "allenai/ai2thor on GitHub (blob)",
     "url": "https://github.com/allenai/ai2thor/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:1712.05474",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s10": {
     "title": "ProcTHOR: Large-Scale Embodied AI Using Procedural Generation",
     "url": "https://arxiv.org/abs/2206.06994",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-06"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "ai2-thor-rearrangement",
   "name": "AI2-THOR Rearrangement",
   "full_name": "AI2-THOR Rearrangement (RoomR)",
   "aliases": [
    "Visual Room Rearrangement",
    "RoomR",
    "AI2-THOR Rearrangement Challenge"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores embodied agents that move objects in simulated rooms to restore a goal state; ran as a public challenge.",
   "summary": {
    "text": "Simulated task where an agent must restore moved objects in a room to their earlier positions and states.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Luca Weihs, Matt Deitke, Aniruddha Kembhavi, Roozbeh Mottaghi; repo allenai/ai2thor-rearrangement",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2021-03 (arXiv v1 2021-03-30)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "CVPR 2021, oral (arXiv comment)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "README: 2023 AI2-THOR Rearrangement Challenge (upgraded AI2-THOR from 5.0.0, new dataset with more balanced easy/hard episodes, opening logic changes). GitHub release v0.6.0 on 2023-04-19; last push 2023-08-15.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "README says the 2023 challenge is hosted at the 'CVPR'22' workshop, likely a typo; challenge page covers 2022 only. Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "mobile-manipulation",
      "long-horizon"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Agent navigates and moves/opens objects to restore a room; capability ids mapped by us. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Abstract iTHOR agent with navigation and object-interaction actions; no named robot. Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "kitchen",
      "home"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "iTHOR floor plans (rooms). Checked 2026-10-10."
    },
    "scale": {
     "value": "120 scenes; 6,000 distinct rearrangement settings (RoomR); 72 object types",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "progress"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Headline metric '% Fixed Strict'; task.metrics() also returns prop_fixed, prop_fixed_strict, energy and num_misplaced. Exact definition of 'strict' not read today. Checked 2026-10-10."
    },
    "top_score": {
     "value": "README: 1-Phase SoTA 'ProcTHOR + Fine-Tuning' 24.47% Fixed Strict (test); 2-Phase 2022 winner MaSS 16.56%; current 2-Phase SoTA TIDEE + open everything 28.94%",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Official (historical): leaderboard.allenai.org/ithor_rearrangement_1phase_2022 and _2phase_2022; host did not resolve (DNS failure) on 2026-10-10)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API licence detection. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No real-robot evaluation found in paper abstract, README or challenge page."
    },
    "citations": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "github_stars": {
     "value": 129,
     "display": "129",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No challenge after 2023 found; last push 2023-08. Checked 2026-10-10."
    },
    "version": {
     "value": "2023 challenge dataset; GitHub release v0.6.0 (2023-04-19)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "license_data": {
     "value": "Dataset splits ship as data/*.pkl.gz in the Apache-2.0 repo; no separate dataset licence found",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Visual Room Rearrangement",
     "url": "https://arxiv.org/abs/2103.16544",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-03"
    },
    "s2": {
     "title": "allenai/ai2thor-rearrangement on GitHub (repository)",
     "url": "https://github.com/allenai/ai2thor-rearrangement",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "https://ai2thor.allenai.org/rearrangement/",
     "url": "https://ai2thor.allenai.org/rearrangement/",
     "type": "site",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "alfred",
   "name": "ALFRED",
   "full_name": "ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks",
   "aliases": [
    "Action Learning From Realistic Environments and Directives",
    "ALFRED Challenge"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores a simulated household agent acting in 3D AI2-THOR scenes from language instructions; fixed splits, metrics and a maintained leaderboard.",
   "summary": {
    "text": "ALFRED is a simulated benchmark in which an agent follows written English instructions to finish multi-step household tasks, such as rinsing a mug and putting it in the coffee maker, in 120 AI2-THOR rooms. The organisers score each entry on a hidden test set, and the headline number is task success in rooms not seen in training.",
    "sources": [
     "s1",
     "s2",
     "s3"
    ],
    "short": "ALFRED is a benchmark in which an AI agent follows written instructions to do household tasks in simulated rooms. Agents are scored on test rooms they did not see in training."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper and the project site present ALFRED as a benchmark with fixed splits, fixed metrics and a test server."
    },
    "kind_secondary": {
     "value": [
      "dataset",
      "challenge"
     ],
     "display": "Also a dataset of 8,055 expert demonstrations with 25,743 written directives, and the task of four workshop challenges: ECCV 2020 and CVPR 2021, 2022 and 2023. The 2023 edition combined ALFRED with TEACh.",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s16",
      "s17",
      "s18",
      "s19"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "2.1.0",
     "display": "Dataset release 2.1.0 (json, json_feat and full packages) for the AI2-THOR 2.1.0 simulator. No later dataset version exists.",
     "level": "verified",
     "sources": [
      "s4",
      "s11",
      "s13"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AI2-THOR 2.1.0",
       "display": "Pinned in requirements.txt together with PyTorch 1.1.0. Uploaded to PyPI on 2019-09-06. The latest AI2-THOR on PyPI is 5.0.0 (2022-12-13).",
       "level": "verified",
       "sources": [
        "s13",
        "s15"
       ]
      },
      {
       "value": "Paper v2 (2020-03-31)",
       "display": "Results table updated after the 2020-03-28 change that selects the target object by mask overlap (IoU).",
       "level": "verified",
       "sources": [
        "s1",
        "s4"
       ]
      },
      {
       "value": "Errata for the Goto sub-goal",
       "display": "The Goto sub-goal success rule changed to 'within 3 planner steps'. The main results and the leaderboard are not affected.",
       "level": "verified",
       "sources": [
        "s10",
        "s4"
       ],
       "note": "Small date conflict: the README change log dates the errata 2020-10-14; the errata file says the rule changed 'as of 14 Dec 2020'."
      },
      {
       "value": "Quickstart data fix (2020-10-26)",
       "display": "A missing stop frame was added to the json_feat package.",
       "level": "verified",
       "sources": [
        "s4",
        "s12"
       ]
      }
     ],
     "short": "Data 2.1.0 for AI2-THOR 2.1.0"
    },
    "publishers": {
     "value": [
      "University of Washington",
      "Carnegie Mellon University",
      "Allen Institute for AI",
      "NVIDIA"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "University of Washington",
       "display": "Paul G. Allen School. Seven of the eight authors list it.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Carnegie Mellon University",
       "display": "Language Technologies Institute (Yonatan Bisk).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Allen Institute for AI",
       "display": "Yonatan Bisk, Winson Han, Roozbeh Mottaghi. The data bucket on Amazon S3 is named ai2-vision-alfred.",
       "level": "verified",
       "sources": [
        "s2",
        "s12"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Second affiliation of Dieter Fox.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "note": "Authors: Mohit Shridhar, Jesse Thomason, Daniel Gordon, Yonatan Bisk, Winson Han, Roozbeh Mottaghi, Luke Zettlemoyer, Dieter Fox."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "From the author affiliations: led from the University of Washington, with AI2 and NVIDIA co-authors."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All listed institutions are in the United States."
    },
    "first_release": {
     "value": "2019-12",
     "display": "arXiv v1 on 2019-12-03. Published at CVPR 2020.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub repository was created on 2019-12-12 (GitHub API).",
     "short": "December 2019. Published at CVPR 2020."
    },
    "latest_update": {
     "value": "2026-09",
     "display": "Newest leaderboard entry on 2026-09-28. Last repository commit on 2026-02-05 (README only). Since April 2025, teams email their results file and the authors post the scores.",
     "level": "verified",
     "sources": [
      "s3",
      "s9",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "September 2026. A new entry was added to the leaderboard."
    },
    "status": {
     "value": "maintained",
     "display": "The leaderboard still receives and posts entries in 2026. The code has not changed since 2021-08 apart from README edits and one dependency pin.",
     "level": "inferred",
     "sources": [
      "s3",
      "s9",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Commit file lists: every commit after 2021-08-29 touches only README.md or requirements.txt (a flask version pin authored 2023-02-28, merged 2025-04-23). The Ai2-hosted leaderboard was deprecated in April 2025 and replaced by email submissions. 23 issues are open (GitHub API).",
     "short": "The leaderboard is active. The code has not changed since 2021."
    },
    "capability": {
     "value": [
      "instruction-following",
      "long-horizon",
      "navigation",
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper describes long, compositional household tasks that mix navigation and object interaction, driven by a written goal and step-by-step instructions. Manipulation is symbolic: the agent chooses an action and marks the target object with a pixel mask; there is no arm or grasp physics."
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "object-instance",
      "language"
     ],
     "display": "The headline 'unseen' test split uses 8 rooms that do not appear in training, with new object variants and newly written instructions. The 'seen' split reuses training rooms.",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper Table 2 and Section 3: unseen validation (4 scenes) and unseen test (8 scenes) are distinct from training and from each other; the split examines generalisation 'to entirely new spaces with novel object class variations'. The 7 task types and 84 object classes are the same in training and test.",
     "short": "New rooms, object variants and instructions"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "AI2-THOR 2.1.0",
     "display": "AI2-THOR 2.1.0, a Unity-based household simulator from the Allen Institute for AI. The paper used AI2-THOR 2.0; the code requires 2.1.0.",
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "AI2-THOR 2.1.0 (Unity)"
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "inferred",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "An egocentric agent with 5 navigation actions (move ahead, rotate left or right, look up or down), 7 interaction actions (pick up, put, open, close, toggle on, toggle off, slice) that target objects through a pixel mask, and a stop action. It has no robot body model. 'virtual-agent' is the closest taxonomy value; 'mobile-manipulator' would overstate the physics. At test time, leaderboard entries may use only RGB images and language."
    },
    "robots": {
     "value": "None",
     "display": "No robot model. An abstract first-person agent with discrete steps and mask-based object interaction.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "kitchen",
      "home"
     ],
     "display": "120 single rooms: 30 kitchens, 30 bathrooms, 30 bedrooms and 30 living rooms. These are single rooms, not whole houses.",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Room counts from the paper, Section 3. Bathrooms, bedrooms and living rooms are mapped to 'home'."
    },
    "tasks": {
     "value": 7,
     "display": "7 task types (for example Pick & Place, Heat & Place, Clean & Place, Examine in Light) over 2,685 combinations of object, receptacle and room",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "7 task types in 2,685 task settings"
    },
    "scenes": {
     "value": 120,
     "display": "120 rooms in AI2-THOR. Splits by room: train 108, validation seen 88, validation unseen 4, test seen 107, test unseen 8.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper Table 2. Seen rooms are subsets of the training rooms.",
     "short": "120 rooms, including 8 unseen test rooms"
    },
    "objects": {
     "value": 84,
     "display": "84 object classes: 58 objects to handle and 26 receptacles. Each class has several visual variants, for example 30 apples.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "demonstrations": {
     "value": 8055,
     "display": "8,055 expert demonstrations (3 per task setting) generated by a PDDL planner, with 25,743 crowd-written English directives. Average 50 steps; 428,322 image-action pairs.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Directives were written on Amazon Mechanical Turk by at least three annotators per demonstration and checked by a second group of workers.",
     "items": [
      {
       "value": "Directive splits",
       "display": "Train 21,023; validation seen 820; validation unseen 821; test seen 1,533; test unseen 1,529 (paper Table 2)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Planner-made, filtered",
       "display": "The organisers' retrospective says task configurations the PDDL planner could not solve were dropped, which leaves fewer corner cases than in TEACh.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "Download sizes",
       "display": "Trajectory JSONs 35,748,602 bytes; JSONs with ResNet features 17,835,841,659 bytes; full package 116,073,781,310 bytes (HTTP Content-Length, read 2026-10-10)",
       "level": "verified",
       "sources": [
        "s12",
        "s11"
       ],
       "note": "The README describes these as 35 MB, about 17 GB and about 109 GB."
      }
     ],
     "short": "8,055 planner demonstrations and 25,743 directives"
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Task success is the headline. Goal-condition success gives partial credit. Both also come in a path-length-weighted form."
    },
    "metric_detail": {
     "value": "unseen test success rate",
     "display": "Task success: an episode counts only if every goal condition holds at the end, for example a potato slice is heated and lies on a counter. Goal-condition success: the share of goal conditions met (2.55 per task on average), which gives partial credit. Path-length-weighted (PLW) versions multiply each score by the expert's step count divided by the larger of the expert's and the agent's step counts, so taking twice as long as the expert halves the credit. The leaderboard ranks entries by success rate on the unseen test split.",
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Paper Section 5.1 and the leaderboard 'Scoring' section. Episodes end after 1,000 steps or 10 failed actions. Humans scored 91.0% task success and 94.5% goal-condition success on 100 unseen test directives (paper Section 6.4).",
     "short": "Success rate on unseen test rooms, with all goals met"
    },
    "trials": {
     "value": "one episode per test directive",
     "display": "Each of the 1,529 unseen and 1,533 seen test directives is run once. Limits: 1,000 steps and 10 failed actions per episode.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Teams run their agent locally on the test instructions and submit the action sequences; the server replays them in the simulator. The README forbids changing the step and failure limits."
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s3",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "The leaderboard shows single numbers. One study reports a seed-to-seed spread of 3 points in success rate for the same model on unseen validation (s20)."
    },
    "evaluator": {
     "value": "organiser-run",
     "display": "The organisers. Teams submit action sequences for the hidden test set; the server replays them in the simulator and computes the metrics.",
     "level": "verified",
     "sources": [
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The agent itself runs on the team's machine, so the organisers check input rules through a questionnaire (RGB and language only at test time) rather than by running the policy."
    },
    "leaderboard": {
     "value": "official",
     "display": "Official leaderboard on askforalfred.com: 95 entries from 2020-03-28 to 2026-09-28. One submission per 7 days; all submissions are public; email submission since April 2025.",
     "level": "verified",
     "sources": [
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Row count and dates from the CSV embedded in the leaderboard page, read 2026-10-10.",
     "short": "Official and active, with 95 entries"
    },
    "top_score": {
     "value": 68.52,
     "display": "68.52% task success on unseen test rooms (GRL, 2025-07-10, team anonymous until acceptance). Humans: 91.0%. All values below are unseen test success rates from the official leaderboard.",
     "level": "verified",
     "sources": [
      "s3",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Organiser-scored numbers. Several top rows are anonymous and have no public paper yet (GRL, PACE-Agent). Entries differ in whether they use the step-by-step instructions (see issues.i3).",
     "items": [
      {
       "value": 0.39,
       "display": "Seq2Seq+PM baseline, 2020-03: 0.39% (seen test 3.98%)",
       "level": "verified",
       "sources": [
        "s3",
        "s2"
       ],
       "data": {
        "model": "Seq2Seq+PM (paper baseline)",
        "date": "2020-03",
        "avg": 0.39,
        "rl": false
       }
      },
      {
       "value": 4.45,
       "display": "ECCV 2020 challenge winner (Nguyen and Okatani, Tohoku University), 2020-08: 4.45%",
       "level": "verified",
       "sources": [
        "s3",
        "s16"
       ],
       "data": {
        "model": "ECCV 2020 winner",
        "date": "2020-08",
        "avg": 4.45,
        "rl": false
       }
      },
      {
       "value": 13.87,
       "display": "HiTUT, 2021-01: 13.87%",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "data": {
        "model": "HiTUT",
        "date": "2021-01",
        "avg": 13.87,
        "rl": false
       }
      },
      {
       "value": 16.29,
       "display": "HLSM ([EAI21] entry, CVPR 2021 winner), 2021-06: 16.29%",
       "level": "verified",
       "sources": [
        "s3",
        "s17",
        "s26"
       ],
       "data": {
        "model": "HLSM",
        "date": "2021-06",
        "avg": 16.29,
        "rl": false
       }
      },
      {
       "value": 26.49,
       "display": "FILM, 2021-09: 26.49%. A later FILM entry (2022-02) scored 27.80%.",
       "level": "verified",
       "sources": [
        "s3",
        "s27"
       ],
       "data": {
        "model": "FILM",
        "date": "2021-09",
        "avg": 26.49,
        "rl": false
       }
      },
      {
       "value": 36.07,
       "display": "EPA ([EAI22] entry, CVPR 2022 winner), 2022-05: 36.07%; path-weighted only 2.92%",
       "level": "verified",
       "sources": [
        "s3",
        "s18"
       ],
       "data": {
        "model": "EPA",
        "date": "2022-05",
        "avg": 36.07,
        "rl": false
       }
      },
      {
       "value": 45.72,
       "display": "Prompter, 2022-08: 45.72%. Its authors report the 45.32% row instead (see issues.i3).",
       "level": "verified",
       "sources": [
        "s3",
        "s23"
       ],
       "data": {
        "model": "Prompter",
        "date": "2022-08",
        "avg": 45.72,
        "rl": false
       }
      },
      {
       "value": 50.36,
       "display": "ECLAIR ([EAI23] entry, CVPR 2023 winner), 2023-06: 50.36%",
       "level": "verified",
       "sources": [
        "s3",
        "s19"
       ],
       "data": {
        "model": "ECLAIR",
        "date": "2023-06",
        "avg": 50.36,
        "rl": false
       }
      },
      {
       "value": 62,
       "display": "RoboGPT, 2023-12: 62.00%; path-weighted 33.61%",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "data": {
        "model": "RoboGPT",
        "date": "2023-12",
        "avg": 62,
        "rl": false
       }
      },
      {
       "value": 62.35,
       "display": "EPO, 2024-02: 62.35%",
       "level": "verified",
       "sources": [
        "s3",
        "s30"
       ],
       "data": {
        "model": "EPO",
        "date": "2024-02",
        "avg": 62.35,
        "rl": false
       }
      },
      {
       "value": 68.52,
       "display": "GRL, 2025-07: 68.52%; path-weighted 45.47%. Highest on the leaderboard.",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "data": {
        "model": "GRL",
        "date": "2025-07",
        "avg": 68.52,
        "rl": false
       }
      },
      {
       "value": 65.53,
       "display": "PACE-Agent, 2026-09: 65.53%. The same team posted 62.52% on 2026-09-10 and 63.83% on 2026-09-21.",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "data": {
        "model": "PACE-Agent",
        "date": "2026-09",
        "avg": 65.53,
        "rl": false
       }
      },
      {
       "value": 91,
       "display": "Human success on 100 unseen test directives: 91.0% (path-weighted 85.8%)",
       "level": "verified",
       "sources": [
        "s2",
        "s3"
       ]
      }
     ],
     "short": "68.52% success on unseen rooms (July 2025)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: MIT, copyright 2019 ALFRED."
    },
    "license_data": {
     "value": "MIT",
     "display": "MIT, by our reading. The README's licence section says 'MIT License'; there is no separate data licence.",
     "level": "inferred",
     "sources": [
      "s4",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "The repository does not say explicitly that MIT covers the data hosted on Amazon S3. We read the repository licence as covering it. The data README and download script name no other terms.",
     "short": "MIT (repository licence)"
    },
    "license_assets": {
     "value": "Apache-2.0",
     "display": "The rooms and 3D objects come with the AI2-THOR simulator, whose repository and PyPI package are Apache-2.0.",
     "level": "inferred",
     "sources": [
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "We found no separate licence for the scenes and object models in AI2-THOR's builds. We read the Apache-2.0 licence of the AI2-THOR repository as covering them. Not checked: third-party asset terms inside the Unity builds."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s11",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Direct downloads from Amazon S3, no registration. The three archive links answered HTTP 200 on 2026-10-10 (headers only). Test-set scores require emailing the results file."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s6",
      "s4",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Code MIT, data MIT by our reading, simulator Apache-2.0. All three allow commercial use with attribution. Asset coverage is inferred (see license_assets). Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No published comparison of ALFRED scores with real-robot results.",
     "level": "inferred",
     "sources": [
      "s2",
      "s22",
      "s37"
     ],
     "checked": "2026-10-10",
     "note": "ALFRED actions are discrete steps and mask-based interactions with no robot model, so a direct robot transfer is not defined. ReALFRED (ECCV 2024) moves ALFRED-style tasks into 3D-captured real homes, still in simulation, and finds that methods built for ALFRED score lower on all metrics. RoboGPT, a top-10 leaderboard entry, reports no real-robot experiment in its v3 text. See searched.",
     "short": "Not checked against real robots"
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "ALFWorld",
      "DialFRED",
      "ALFRED-L",
      "ReALFRED",
      "EB-ALFRED (EmbodiedBench)",
      "Behavior-IL / Environment-IL",
      "LoTa-Bench"
     ],
     "display": "Benchmarks built on ALFRED tasks, scenes or data",
     "level": "verified",
     "sources": [
      "s32",
      "s33",
      "s21",
      "s22",
      "s34",
      "s35",
      "s36"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ALFWorld",
       "display": "2020-10, ICLR 2021. Text-game versions of ALFRED tasks (TextWorld) linked to the visual environment. University of Washington and Microsoft Research.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "DialFRED",
       "display": "2022-02, RA-L. Adds questions and answers: the agent may ask a human for help. 53K question-answer pairs.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "ALFRED-L",
       "display": "2022, EMNLP. A test split with changed task structures to check whether agents follow the language (see issues.i2).",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "ReALFRED",
       "display": "2024-07, ECCV 2024. ALFRED-style tasks in larger, 3D-captured multi-room scenes (see issues.i5).",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "EB-ALFRED",
       "display": "2025-02, ICML 2025. ALFRED-based environment inside EmbodiedBench for multimodal language models.",
       "level": "verified",
       "sources": [
        "s34"
       ]
      },
      {
       "value": "Behavior-IL / Environment-IL",
       "display": "2024-03, ICLR 2024. Continual-learning setups on ALFRED.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "LoTa-Bench",
       "display": "2024-02, ICLR 2024. Tests language-model task planners on ALFRED in AI2-THOR (and on Watch-And-Help).",
       "level": "verified",
       "sources": [
        "s36"
       ]
      }
     ],
     "note": "TEACh (Amazon) uses the same simulator and was paired with ALFRED in the 2023 challenge, but it is a separate dataset. Not a complete list.",
     "short": "7 benchmarks built on it"
    },
    "citations": {
     "value": 1230,
     "display": "1,230 (Semantic Scholar; 179 influential)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "1,230"
    },
    "github_stars": {
     "value": 531,
     "display": "531 stars, 107 forks (askforalfred/alfred)",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "531"
    },
    "used_by": {
     "value": "95 leaderboard entries from 2020-03 to 2026-09, including 13 entries dated 2025 or 2026. The README lists 10 open-source models that beat the paper baseline.",
     "level": "verified",
     "sources": [
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Counted from the leaderboard CSV. Many entries are repeat submissions by the same team.",
     "items": [
      {
       "value": "Episodic Transformer (E.T.)",
       "display": "Inria and Google Research, ICCV 2021",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "HLSM",
       "display": "NVIDIA, Cornell and others, CoRL 2021. CVPR 2021 challenge winner.",
       "level": "verified",
       "sources": [
        "s26",
        "s17"
       ]
      },
      {
       "value": "FILM",
       "display": "Carnegie Mellon University and Facebook AI Research, ICLR 2022",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "EPA",
       "display": "ServiceNow Research and Queen's University. CVPR 2022 challenge winner.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "Prompter",
       "display": "Hitachi, 2022",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "LLM-Planner",
       "display": "The Ohio State University, 2022. Few-shot planning with a large language model.",
       "level": "verified",
       "sources": [
        "s28",
        "s4"
       ]
      },
      {
       "value": "CAPEAM",
       "display": "ICCV 2023. Listed by the README as an open-source model.",
       "level": "verified",
       "sources": [
        "s29",
        "s4"
       ]
      },
      {
       "value": "EPO",
       "display": "Brown University, EMNLP 2024. 62.35% on the leaderboard.",
       "level": "verified",
       "sources": [
        "s30",
        "s3"
       ]
      },
      {
       "value": "EmBERT",
       "display": "Heriot-Watt University and Amazon Alexa AI, 2021",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ],
     "short": "95 leaderboard entries (2020 to 2026)"
    },
    "industry_use": {
     "value": [
      "Amazon",
      "Hitachi",
      "ServiceNow",
      "NVIDIA",
      "Google",
      "Meta",
      "Microsoft"
     ],
     "level": "verified",
     "sources": [
      "s31",
      "s20",
      "s21",
      "s23",
      "s18",
      "s26",
      "s25",
      "s27",
      "s32"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Amazon",
       "display": "Amazon Alexa AI co-authored EmBERT, the ALFRED-L study and the validation-set study.",
       "level": "verified",
       "sources": [
        "s31",
        "s21",
        "s20"
       ]
      },
      {
       "value": "Hitachi",
       "display": "Prompter, a top-10 entry in 2022.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "ServiceNow",
       "display": "ServiceNow Research co-built EPA, the CVPR 2022 challenge winner.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Co-author of ALFRED; HLSM (CVPR 2021 winner) lists NVIDIA first among its affiliations.",
       "level": "verified",
       "sources": [
        "s2",
        "s26"
       ]
      },
      {
       "value": "Google",
       "display": "Google Research co-authored E.T.; the first author of ALFRED-L was at Google AI.",
       "level": "verified",
       "sources": [
        "s25",
        "s21"
       ]
      },
      {
       "value": "Meta",
       "display": "Facebook AI Research co-authored FILM.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "Microsoft",
       "display": "Microsoft Research co-built ALFWorld, the text version of ALFRED.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      }
     ],
     "note": "Company research groups publishing results or derived benchmarks. We found no product or frontier-model report that cites an ALFRED leaderboard score."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "Validation gains did not carry over to the test set",
     "text": "Kim et al. (UNC Chapel Hill and Amazon Alexa AI, 2022) built a model that beat the best published models on unseen validation (13.8% success against 12.55% for ABP) but scored lower on unseen test (8.57% against 15.43%). They saw a 3-point spread in validation success across random seeds for the same model. The unseen validation split has only 4 rooms and the unseen test split 8 rooms, so model selection on one small set of rooms may not transfer to the other. The authors suggest averaging several runs of each model when ranking methods.",
     "level": "verified",
     "sources": [
      "s20",
      "s2"
     ],
     "status": "open",
     "short": "A model that led on the validation rooms fell behind on the test rooms."
    },
    {
     "id": "i2",
     "type": "shortcut",
     "title": "Agents make little use of the step-by-step instructions",
     "text": "ALFRED-L (EMNLP 2022) tested six models from 2020 and 2021. Removing every step-by-step instruction and keeping only the goal lowered success by at most 6.7% in relative terms. Adding one 'go back' step to the instructions cut the success of the best model (ET+Synth) on seen rooms from 44.7% to 9.2%. The authors conclude that the models rely on the usual order in which objects are visited in ALFRED's 7 task structures. Newer modular and language-model agents were not tested.",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "status": "open",
     "short": "Removing the step-by-step instructions barely changed the success of older models."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "Leaderboard rows use different inputs",
     "text": "The leaderboard ranks all entries by one success rate, whether or not they use the step-by-step instructions. Some rows are labelled goal-only ('HIA-High-Goal-Only', 'high level only'). Prompter's paper reports 45.32% with step-by-step instructions and 41.53% with the goal alone. Its leaderboard row of 45.72% used domain knowledge that its authors judged 'too specific to ALFRED', so the paper reports the lower 45.32% row ('Prompter, no slice replay'). Teams answer a questionnaire on their inputs, but the answers are not shown in the table.",
     "level": "verified",
     "sources": [
      "s3",
      "s23",
      "s4"
     ],
     "status": "open",
     "short": "The leaderboard ranks entries that use only the goal together with entries that use the full instructions."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The benchmark depends on an old simulator release",
     "text": "The code requires AI2-THOR 2.1.0 (uploaded to PyPI on 2019-09-06), PyTorch 1.1.0 and an X server for rendering. The latest AI2-THOR on PyPI is 5.0.0 (2022-12-13). Results depend on replaying actions in version 2.1.0, so new work cannot move to newer AI2-THOR releases without breaking comparability.",
     "level": "inferred",
     "sources": [
      "s13",
      "s15",
      "s4"
     ],
     "status": "open",
     "note": "The comparability point is our inference from the replay-based scoring.",
     "short": "Scores depend on AI2-THOR 2.1.0, a simulator release from 2019."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Scores drop in more realistic scenes",
     "text": "ReALFRED (ECCV 2024) rebuilds ALFRED-style tasks in larger, multi-room, 3D-captured scenes. Methods designed for ALFRED consistently scored lower on all metrics there. ALFRED's own rooms are single game-engine rooms.",
     "level": "verified",
     "sources": [
      "s22",
      "s2"
     ],
     "status": "open",
     "short": "Methods built for ALFRED score lower in multi-room scenes captured in 3D."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "An ALFRED score shows how well an agent completes scripted household tasks from written instructions in 120 game-engine rooms. It is weak evidence about robots. Actions are discrete steps, objects are selected by drawing a mask, and no study has compared ALFRED scores with real-robot results.",
     "basis": [
      "facts.embodiment",
      "facts.sim_to_real",
      "facts.metric_detail",
      "issues.i5"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A good ALFRED score is weak evidence about how an agent would do on real robots."
    },
    {
     "id": "r2",
     "text": "Small gaps between top entries are hard to interpret. The unseen test split covers 8 rooms, and entries differ in which instructions they use. In one study, a model that led on the validation rooms scored lower on the test rooms.",
     "basis": [
      "facts.scenes",
      "issues.i1",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Do not read much into gaps of a few points between top entries."
    },
    {
     "id": "r3",
     "text": "Read the path-weighted score (a score that gives less credit when the agent takes more steps than the expert) next to the success rate. Several top entries reach their success rate through long searches. GRL scores 68.52% success but 45.47% path-weighted, and EPA scores 36.07% against 2.92%.",
     "basis": [
      "facts.top_score",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the path-weighted scores as well. Top agents often take long routes to finish a task."
    },
    {
     "id": "r4",
     "text": "ALFRED is still an active comparison point for language-driven household agents. The leaderboard received new entries in September 2026, and the best entry (68.52%) remains well below human success (91.0%).",
     "basis": [
      "facts.leaderboard",
      "facts.top_score",
      "facts.status"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "ALFRED is still in active use. The best entry is well below human performance."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well an agent would do on a real robot.",
     "sub": "We found no study that compares ALFRED scores with real-robot results.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether an agent can physically grasp objects.",
     "sub": "The agent selects an object by drawing a mask, which marks the object in the camera image.",
     "basis": [
      "facts.embodiment",
      "facts.capability"
     ]
    },
    {
     "id": "l3",
     "text": "Whether an agent follows each step of the instructions.",
     "sub": "Older models ignored the step-by-step instructions.",
     "basis": [
      "issues.i2"
     ]
    },
    {
     "id": "l4",
     "text": "How efficiently an agent completes a task.",
     "sub": "The leaderboard ranks entries by success rate. It does not use path length.",
     "basis": [
      "facts.metric_detail",
      "facts.top_score"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "ALFRED paper (arXiv 1912.01734v2, full text), project site and challenge pages (EVAL 2020, EAI21, EAI22, EAI23), README; web searches on 2026-10-10 for 'ALFRED real robot', 'ALFRED sim-to-real' and for real-robot experiments by top leaderboard entries; RoboGPT v3 text (no physical robot); ReALFRED (real scans, still simulated). No paired evaluation of the same agents on ALFRED and on a real robot found.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data",
     "where": "README licence section, LICENSE file, data README, download_data.sh, project site. No separate data licence found.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "AI2-THOR repository LICENSE and PyPI metadata. No separate licence for scenes or 3D objects found; third-party terms inside the Unity builds not checked.",
     "date": "2026-10-10"
    },
    {
     "for": "uncertainty_reported",
     "where": "Leaderboard table (single numbers), paper Table 3 (single numbers), s20 (seed spread).",
     "date": "2026-10-10"
    },
    {
     "for": "industry_use (frontier-model reports)",
     "where": "Web search for ALFRED and EB-ALFRED in 2025-2026 model technical reports. None found that cites an ALFRED leaderboard score.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks (arXiv abstract page)",
     "url": "https://arxiv.org/abs/1912.01734",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2019-12",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "ALFRED paper, full text v2 (Tables 2 and 3, Sections 3, 5 and 6)",
     "url": "https://arxiv.org/pdf/1912.01734",
     "type": "paper",
     "publisher": "arXiv (CVPR 2020 version)",
     "date": "2020-03",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "ALFRED leaderboard (rules, metrics and the results table embedded as CSV)",
     "url": "https://askforalfred.com/leaderboard/leaderboard.html",
     "type": "leaderboard",
     "publisher": "ALFRED authors",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "ALFRED GitHub README (leaderboard rules, April 2025 update, change log, licence)",
     "url": "https://github.com/askforalfred/alfred/blob/master/README.md",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "ALFRED project site",
     "url": "https://askforalfred.com/",
     "type": "site",
     "publisher": "ALFRED authors",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "ALFRED LICENSE file",
     "url": "https://github.com/askforalfred/alfred/blob/master/LICENSE",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2019",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "Semantic Scholar API record for arXiv:1912.01734",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:1912.01734?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "GitHub API: askforalfred/alfred (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/askforalfred/alfred",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "ALFRED commit history (files changed per commit)",
     "url": "https://github.com/askforalfred/alfred/commits/master",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2026-02-05",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "ALFRED errata for Goto sub-goal evaluation",
     "url": "https://github.com/askforalfred/alfred/blob/master/models/ERRATA.md",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "ALFRED dataset README and download script",
     "url": "https://github.com/askforalfred/alfred/blob/master/data/README.md",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "ALFRED data package on Amazon S3 (HTTP headers of json, json_feat and full 2.1.0 archives)",
     "url": "https://ai2-vision-alfred.s3-us-west-2.amazonaws.com/json_feat_2.1.0.7z",
     "type": "dataset",
     "publisher": "Allen Institute for AI (S3 bucket ai2-vision-alfred)",
     "date": "2020-10-26",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "ALFRED requirements.txt (ai2thor==2.1.0, torch==1.1.0)",
     "url": "https://github.com/askforalfred/alfred/blob/master/requirements.txt",
     "type": "repo",
     "publisher": "askforalfred",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "AI2-THOR repository LICENSE (Apache License 2.0)",
     "url": "https://github.com/allenai/ai2thor/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2017",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "PyPI: ai2thor (release history; 2.1.0 uploaded 2019-09-06; latest 5.0.0)",
     "url": "https://pypi.org/project/ai2thor/",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2022-12-13",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "ALFRED Challenge at the EVAL workshop, ECCV 2020",
     "url": "https://askforalfred.com/EVAL/",
     "type": "site",
     "publisher": "ALFRED authors",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "ALFRED Challenge at the Embodied AI Workshop, CVPR 2021",
     "url": "https://askforalfred.com/EAI21/",
     "type": "site",
     "publisher": "ALFRED authors",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "ALFRED Challenge at the Embodied AI Workshop, CVPR 2022",
     "url": "https://askforalfred.com/EAI22/",
     "type": "site",
     "publisher": "ALFRED authors",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "ALFRED+TEACh Generalist Language Grounding Agents Challenge, CVPR 2023",
     "url": "https://askforalfred.com/EAI23/",
     "type": "site",
     "publisher": "ALFRED and TEACh authors",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "On the Limits of Evaluating Embodied Agent Model Generalization Using Validation Sets (Kim et al., ACL 2022 Insights Workshop)",
     "url": "https://arxiv.org/abs/2205.09249",
     "type": "paper",
     "publisher": "arXiv (UNC Chapel Hill, Amazon Alexa AI)",
     "date": "2022-05",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "ALFRED-L: Investigating the Role of Language for Action Learning in Interactive Visual Environments (EMNLP 2022)",
     "url": "https://aclanthology.org/2022.emnlp-main.636.pdf",
     "type": "paper",
     "publisher": "EMNLP 2022 (Google AI, Amazon Alexa AI, UNC, USC)",
     "date": "2022-12",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "ReALFRED: An Embodied Instruction Following Benchmark in Photo-Realistic Environments (ECCV 2024)",
     "url": "https://www.ecva.net/papers/eccv_2024/papers_ECCV/papers/01610.pdf",
     "type": "paper",
     "publisher": "ECCV 2024 (Seoul National University, Yonsei University)",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Prompter: Utilizing Large Language Model Prompting for a Data Efficient Embodied Instruction Following (Table I and footnote 2)",
     "url": "https://arxiv.org/abs/2211.03267",
     "type": "paper",
     "publisher": "arXiv (Hitachi)",
     "date": "2022-11",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Retrospectives on the Embodied AI Workshop (Sections 3.3.2 and 3.3.3)",
     "url": "https://arxiv.org/abs/2210.06849",
     "type": "paper",
     "publisher": "arXiv (Embodied AI Workshop organisers)",
     "date": "2022-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Episodic Transformer for Vision-and-Language Navigation (E.T.)",
     "url": "https://arxiv.org/abs/2105.06453",
     "type": "paper",
     "publisher": "ICCV 2021 (Inria, Google Research, Brown University)",
     "date": "2021-05",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "A Persistent Spatial Semantic Representation for High-level Natural Language Instruction Execution (HLSM)",
     "url": "https://arxiv.org/abs/2107.05612",
     "type": "paper",
     "publisher": "CoRL 2021 (NVIDIA, Cornell, University of Washington, University of Toronto)",
     "date": "2021-07",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "FILM: Following Instructions in Language with Modular Methods",
     "url": "https://arxiv.org/abs/2110.07342",
     "type": "paper",
     "publisher": "ICLR 2022 (Carnegie Mellon University, Facebook AI Research)",
     "date": "2021-10",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "LLM-Planner: Few-Shot Grounded Planning for Embodied Agents with Large Language Models",
     "url": "https://arxiv.org/abs/2212.04088",
     "type": "paper",
     "publisher": "arXiv (The Ohio State University, DEVCOM ARL)",
     "date": "2022-12",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Context-Aware Planning and Environment-Aware Memory for Instruction Following Embodied Agents (CAPEAM)",
     "url": "https://arxiv.org/abs/2308.07241",
     "type": "paper",
     "publisher": "ICCV 2023",
     "date": "2023-08",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "EPO: Hierarchical LLM Agents with Environment Preference Optimization",
     "url": "https://arxiv.org/abs/2408.16090",
     "type": "paper",
     "publisher": "EMNLP 2024 (Brown University)",
     "date": "2024-08",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Embodied BERT: A Transformer Model for Embodied, Language-guided Visual Task Completion (EmBERT)",
     "url": "https://arxiv.org/abs/2108.04927",
     "type": "paper",
     "publisher": "arXiv (Heriot-Watt University, Amazon Alexa AI, USC)",
     "date": "2021-08",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "ALFWorld: Aligning Text and Embodied Environments for Interactive Learning",
     "url": "https://arxiv.org/abs/2010.03768",
     "type": "paper",
     "publisher": "ICLR 2021 (University of Washington, Microsoft Research, CMU)",
     "date": "2020-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "DialFRED: Dialogue-Enabled Agents for Embodied Instruction Following",
     "url": "https://arxiv.org/abs/2202.13330",
     "type": "paper",
     "publisher": "RA-L 2022",
     "date": "2022-02",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents",
     "url": "https://arxiv.org/abs/2502.09560",
     "type": "paper",
     "publisher": "ICML 2025",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Online Continual Learning For Interactive Instruction Following Agents (Behavior-IL and Environment-IL on ALFRED)",
     "url": "https://arxiv.org/abs/2403.07548",
     "type": "paper",
     "publisher": "ICLR 2024",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "LoTa-Bench: Benchmarking Language-oriented Task Planners for Embodied Agents",
     "url": "https://arxiv.org/abs/2402.08178",
     "type": "paper",
     "publisher": "ICLR 2024",
     "date": "2024-02",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "RoboGPT: an LLM-based Embodied Long-term Decision Making agent for Instruction Following Tasks (v3)",
     "url": "https://arxiv.org/html/2311.15649v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-11",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Full entry. Re-checked every basic fact at the source. Corrections: capability note (manipulation is mask-based); evaluator set to organiser-run (server replay of submitted actions); sim_to_real recorded as 'none-found' (inferred) instead of unknown. Added leaderboard history from the embedded CSV (95 rows), five issues, derived benchmarks, adoption and licences for the simulator."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "asimov",
   "name": "ASIMOV",
   "full_name": "Generating Robot Constitutions & Benchmarks for Semantic Safety (ASIMOV v1); Can AI Perceive Physical Danger and Intervene? (ASIMOV-2.0)",
   "aliases": [
    "ASIMOV Benchmark",
    "ASIMOV-1.0",
    "ASIMOV-2.0",
    "ASIMOV-Injury",
    "ASIMOV-Multimodal",
    "ASIMOV-Dilemmas",
    "ASIMOV-2.0-Video",
    "ASIMOV-2.0-Constraints"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "An embodied safety benchmark built for the AI models that direct robots. The scope rule includes embodied safety benchmarks.",
   "summary": {
    "text": "ASIMOV is a set of question sets from Google DeepMind that test whether AI models judge physical danger the way people do, using text scenarios drawn from US hospital injury reports and AI-generated images and videos. Scores are the share of answers that match human raters; no robot moves, and the newer agentic variant (ASIMOV-Agentic, 2026) is a separate entry.",
    "short": "ASIMOV is a set of question sets from Google DeepMind. It tests whether AI models recognise physical danger the way human raters do.",
    "sources": [
     "s1",
     "s5",
     "s15"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Both papers call it a benchmark. v1 also describes large training sets for building robot constitutions (see kind_secondary)."
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "v1 also lists training sets of about 2.9 million instructions, most of which were not released.",
     "level": "verified",
     "sources": [
      "s1",
      "s19"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "2.0",
     "display": "Two versions with different questions. v1 (March 2025): Multimodal, Injury and Dilemmas sets. ASIMOV-2.0 (September 2025): Injury (text), Video and Constraints (images). Each released dataset is TFDS version 0.1.0.",
     "level": "verified",
     "sources": [
      "s14",
      "s15",
      "s1",
      "s5",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "The project root redirects to /v1/ or /v2/. Both versions stay downloadable. ASIMOV-Agentic (Google DeepMind, 2026-07) extends the family and has its own Atlas entry.",
     "items": [
      {
       "value": "v1 (ASIMOV-1.0)",
       "display": "Five subsets: Multimodal-Auto, Multimodal-Manual, Injury, Dilemmas-Auto, Dilemmas-SciFi. Plus a 7-item RoboPAIR validation set.",
       "level": "verified",
       "sources": [
        "s1",
        "s19"
       ]
      },
      {
       "value": "ASIMOV-2.0",
       "display": "Three components: Injury (319 text scenarios), Video (287 generated videos), Constraints (164 image and constraint pairs).",
       "level": "verified",
       "sources": [
        "s5"
       ]
      }
     ],
     "short": "v1 (March 2025) and 2.0 (September 2025)"
    },
    "publishers": {
     "value": [
      "Google DeepMind",
      "Princeton University"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s5",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "v1 authors: Sermanet, Majumdar, Irpan, Kalashnikov, Sindhwani. 2.0 authors: Jindal, Kalashnikov, Hofer, Chang, Garikapati, Majumdar, Sermanet, Sindhwani. All list Google DeepMind (Robotics); Anirudha Majumdar also lists Princeton University.",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "All authors of both papers.",
       "level": "verified",
       "sources": [
        "s1",
        "s5"
       ]
      },
      {
       "value": "Princeton University",
       "display": "Second affiliation of Anirudha Majumdar.",
       "level": "verified",
       "sources": [
        "s5",
        "s15"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "All authors are at Google DeepMind."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Google DeepMind works across several countries; the papers give no team location. One co-author is at Princeton (US)."
    },
    "first_release": {
     "value": "2025-03",
     "display": "v1 data uploaded 2025-03-06; arXiv v1 on 2025-03-11. Published at CoRL 2025 (PMLR volume 305, pages 4767 to 4823).",
     "level": "verified",
     "sources": [
      "s19",
      "s2",
      "s3",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "TFDS registered the v1 datasets on 2025-03-07 (tensorflow/datasets commit 'Release of ASIMOV datasets').",
     "short": "March 2025, at CoRL 2025"
    },
    "latest_update": {
     "value": "2025-11",
     "display": "ASIMOV-2.0 paper v2 on 2025-11-21. The 2.0 data was uploaded 2025-09-24 and has not changed since.",
     "level": "verified",
     "sources": [
      "s6",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The 2025-11 revision replaced one results figure (see issues.i3). ASIMOV-2.0 has no conference venue that we found; its arXiv record carries report number GDM-01-01.",
     "short": "November 2025. The paper was revised."
    },
    "status": {
     "value": "maintained",
     "display": "No data change since September 2025. Google still cited it in July 2026 and released an agentic extension.",
     "level": "inferred",
     "sources": [
      "s19",
      "s6",
      "s28",
      "s29",
      "s16",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Data unchanged since 2025-09-24; last paper revision 2025-11-21. The code repository has had no commit since 2025-03-11, and its two issues (2025-03-12, 2025-06-22) have no reply. The Gemini Robotics ER 2 model card (2026-07-30) links its semantic-safety claim to the ASIMOV v2 page.",
     "short": "No updates since November 2025. Still cited."
    },
    "capability": {
     "value": [
      "safety",
      "embodied-reasoning"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "v1: whether an instruction is desirable in context, with or without a written robot constitution. 2.0: injury risk type and severity, effect of an action, when to step in, and following a robot's physical limits."
    },
    "generalisation": {
     "value": [
      "none-stated"
     ],
     "display": "Each item is a new scenario. The benchmark does not define held-out conditions.",
     "level": "inferred",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "v1 says its evaluation sets are disjoint from its training sets for images, contexts, instructions and rules; that split serves constitution learning, not a generalisation test.",
     "short": "Not described"
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Models answer questions about text, images or videos. No robot or simulator runs."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "display": "No robot under test. Some prompts tell the model it is a robot, for example 'I am a humanoid robot'.",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "v1 images start from RoboVQA frames taken by single-arm mobile robots and by people wearing head cameras, then edited with Imagen 3.",
     "short": "None. The test uses questions only."
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "display": "Homes, offices, kitchens, workplaces and science-fiction dilemmas.",
     "level": "inferred",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Injury scenarios come from US emergency-room reports (NEISS 2023). Constraint images show objects with physical properties such as heat, weight or fragility."
    },
    "tasks": {
     "value": "8 question sets",
     "display": "5 v1 subsets and 3 ASIMOV-2.0 components",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "8 question sets"
    },
    "scale": {
     "value": "v1: 2,273 validation instructions; 2.0: 770 items",
     "display": "v1 validation: 310 contexts or images and 2,273 instructions, 1,140 with human labels. v1 in all: 513,679 contexts and 2,942,060 instructions. 2.0: 319 + 287 + 164 = 770 evaluation items.",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v1 Table 1",
       "display": "Validation: Multimodal-Auto 1,311 instructions (789 human labels), Multimodal-Manual 159 (0), Injury 319 (163), Dilemmas-Auto 200 (35), Dilemmas-SciFi 284 (153).",
       "level": "verified",
       "sources": [
        "s1",
        "s4"
       ]
      },
      {
       "value": "v1 released files",
       "display": "Public record counts: injury_val 304, dilemmas_auto_val 34, dilemmas_scifi_val 51, dilemmas_scifi_train 9,004, multimodal_auto_val 50, multimodal_manual_val 59, multimodal_robopair_val 7.",
       "level": "verified",
       "sources": [
        "s19",
        "s20"
       ],
       "note": "Read from each dataset_info.json. Units differ between files (one record per instruction, per context or per image), so some differences from Table 1 may be units; injury_val holds one instruction per record, 304 against 319 in Table 1 (issues.i4)."
      },
      {
       "value": "ASIMOV-2.0",
       "display": "Injury 319 text scenarios; Video 287 videos (193 without a realistic injury, 94 with one); Constraints 164 image and constraint pairs in 8 categories.",
       "level": "verified",
       "sources": [
        "s5",
        "s37"
       ],
       "note": "Released files match: asimov_v2_injuries 319, asimov_v2_videos 287, both constraints files 164."
      }
     ],
     "short": "2,273 test items in v1 and 770 in v2"
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Agreement with human labels. 2.0 also measures timing error in seconds and a constraint violation rate, which have no taxonomy value."
    },
    "metric_detail": {
     "value": "agreement with human labels",
     "display": "v1: 'alignment rate', the share of yes-or-no desirability answers that match human labels, in a normal mode and in an 'adversary' mode where the model is told to flip good and bad. 2.0 Injury: accuracy on four multiple-choice questions (risk type, risk severity, effect of the action, severity after the action). 2.0 Video: risk accuracy, plus error in seconds for the last moment an intervention could prevent injury. 2.0 Constraints: share of answers that point at an object breaking the stated robot limit.",
     "level": "verified",
     "sources": [
      "s1",
      "s5",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "v1's headline 84.3% is the average of normal-mode and adversary-mode alignment. 2.0 labels: 5 raters per item; low-consensus items dropped (Injury), at least 60% consensus and a timestamp spread under 1.0 s (Video), at least 80% consensus (Constraints). No agreement statistic between raters is reported.",
     "short": "Share of answers that match human raters"
    },
    "trials": {
     "value": "one answer per item",
     "display": "Each model answers each item once; the papers do not report repeated runs.",
     "level": "inferred",
     "sources": [
      "s1",
      "s5",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "Not stated explicitly. VLESA (2026-06) applied the official consensus rule and kept 189 of the 287 videos with valid intervention labels for timing metrics.",
     "short": "One answer per item"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s1",
      "s5",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Results are single numbers or plotted points without error bars. The only intervals in the 2.0 paper are for a pointing test in its appendix."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "Data and prompts are public; each team scores its own models."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "No results table on the v1 or v2 project pages or in the repository."
    },
    "top_score": {
     "value": "no single headline score",
     "display": "Each question set has its own score. Text risk questions are near the ceiling (0.94 to 0.96 for Gemini 2.5 Pro). Video and constraint questions are not: the best frontier model still broke stated robot limits in 38.6% of Constraints answers (September 2025).",
     "level": "inferred",
     "sources": [
      "s1",
      "s5",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "2.0 values below are read by us from the paper's plotted figures (Figures 4 to 6); numbers printed in the text are marked as such. All results are self-reported, mostly by Google.",
     "items": [
      {
       "value": "84.3%",
       "display": "v1, 2025-03: best alignment, Gemini 1.5 Pro with a generated 128-line constitution (87.7% normal, 80.9% adversary mode). Without a constitution: 83.6% normal, 33.6% adversary.",
       "level": "verified",
       "sources": [
        "s1",
        "s4"
       ]
      },
      {
       "value": "94.67%",
       "display": "v1 Injury validation, 2025-03: Gemini 1.5 Pro with Robot-Constitution-6. GPT-4-Turbo without a constitution: 90.61%.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      },
      {
       "value": "0.94 / 0.96",
       "display": "2.0 Injury, 2025-09: risk type and severity accuracy, Gemini 2.5 Pro. Text: GPT-5, Gemini 2.5 Pro and Claude Opus 4.1 average 92.3% on risk type.",
       "level": "verified",
       "sources": [
        "s5",
        "s10"
       ]
      },
      {
       "value": "0.74 / 0.66",
       "display": "2.0 Injury, 2025-09: best action-effect accuracy (Claude Opus 4.1) and best post-action risk accuracy (Claude Sonnet 4 and Gemini 2.5 Flash). The text says top models score 74% and 66%.",
       "level": "verified",
       "sources": [
        "s5",
        "s11"
       ]
      },
      {
       "value": "0.84",
       "display": "2.0 Video, 2025-11 version: risk accuracy for Gemini 2.5 Pro (0.89 in the September 2025 version; see issues.i3).",
       "level": "verified",
       "sources": [
        "s9",
        "s8"
       ]
      },
      {
       "value": "56% within 0.5 s",
       "display": "2.0 Video, 2025-09: Gemini 2.5 Pro places the last useful intervention within 0.75 s of human raters on average, and within 0.5 s on 56% of videos (paper text). GPT-5: about 9% (figure).",
       "level": "verified",
       "sources": [
        "s5",
        "s12"
       ]
      },
      {
       "value": "38.6% violations",
       "display": "2.0 Constraints, 2025-09: lowest violation rate among frontier models, Claude Opus 4.1. Highest: Gemini 2.5 Flash-Lite at 75.3%.",
       "level": "verified",
       "sources": [
        "s5",
        "s13"
       ]
      },
      {
       "value": "68.4% satisfied",
       "display": "2.0 Constraints, 2025-10: Gemini Robotics-ER 1.5 fine-tuned on 200 pairs made with the same recipe (31.6% violations). GPT-5: 59.4%.",
       "level": "verified",
       "sources": [
        "s26",
        "s27",
        "s5"
       ],
       "note": "See issues.i2: the training pairs come from the benchmark's own generator."
      }
     ],
     "short": "Text questions are close to the maximum. Video and constraint questions are not."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "asimov-benchmark/code holds a 6-byte README and one notebook, with no LICENSE file; GitHub reports no licence. The TFDS loader scripts in tensorflow/datasets are Apache-2.0, which covers those scripts only."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No licence on the v1 or v2 project pages, in the repository, in any dataset_info.json (no licence field) or in the TFDS builder code. The CC BY 4.0 notice on TFDS catalog pages covers the page content. Sources inside the data (NEISS reports, RoboVQA frames, Imagen and Veo outputs) were not checked for terms."
    },
    "license_assets": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Images and videos are AI-generated (Imagen 3, Veo 3) or edited from RoboVQA frames. No separate licence stated; looked in the same places as license_data."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s17",
      "s19",
      "s15",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Public Google Cloud bucket gs://gresearch/robotics/asimov_*, loadable through TensorFlow Datasets; the v2 page links a Colab notebook. No registration.",
     "short": "Open, through Google Cloud and TensorFlow Datasets (TFDS)"
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s16",
      "s20",
      "s21",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "No licence is stated for the code, data or images. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No study links ASIMOV scores to how a robot behaves. ASIMOV-Agentic (separate entry) has one real-robot stopping test.",
     "level": "inferred",
     "sources": [
      "s1",
      "s5",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "The scores are already agreement with human judges, so the open question is whether answering well predicts acting safely. No paired study found (see searched).",
     "short": "Not checked against robot behaviour"
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Offline questions; no robot runs."
    },
    "citations": {
     "value": 43,
     "display": "43 for v1 (Semantic Scholar; 6 influential); 15 for ASIMOV-2.0 (2 influential)",
     "level": "verified",
     "sources": [
      "s23",
      "s24"
     ],
     "checked": "2026-10-10",
     "short": "43 for v1 and 15 for 2.0"
    },
    "github_stars": {
     "value": 27,
     "display": "27 stars, 5 forks (asimov-benchmark/code, a single notebook)",
     "level": "verified",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "short": "27"
    },
    "used_by": {
     "value": "Google reports ASIMOV results for its own models; two outside groups had published results on ASIMOV-2.0 data by 2026-10. Our count from Semantic Scholar's 43 and 15 citing papers; not exhaustive.",
     "level": "inferred",
     "sources": [
      "s23",
      "s24",
      "s25",
      "s26",
      "s28",
      "s31",
      "s32",
      "s33"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Gemini Robotics",
       "display": "Google DeepMind, 2025-03. ASIMOV-Multimodal and ASIMOV-Injury results for Gemini 2.0 Flash and Gemini Robotics-ER, which is post-trained on such instances.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Gemini Robotics 1.5",
       "display": "Google DeepMind, 2025-10. ASIMOV-2.0 results: Gemini Robotics-ER 1.5 risk (text) 90.0, action (text) 76.0, risk (video) 62.0, intervention (video) 88.4.",
       "level": "verified",
       "sources": [
        "s26",
        "s27"
       ]
      },
      {
       "value": "Gemini Robotics ER 2 model card",
       "display": "Google DeepMind, 2026-07-30. Claims gains on 'semantic safety benchmarks', linked to the ASIMOV v2 page; no ASIMOV numbers in the card text.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "VLESA",
       "display": "Carnegie Mellon, MERL and Harvard, 2026-06. Evaluates on ASIMOV-2.0-Video (189 videos with valid labels).",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "Visual Grounding Safety in Vision-Language Models",
       "display": "UC Riverside and Apple, 2026-10. Turns the 164 ASIMOV-2.0-Constraints items into harmful requests to test refusals.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "What the Guard Misses, the Robot Executes",
       "display": "2026-10. Uses 71 harmless robot instructions from ASIMOV to calibrate safety filters.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      }
     ],
     "short": "Mostly Google. 2 papers from other groups report results."
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Apple",
      "Mitsubishi Electric Research Laboratories"
     ],
     "level": "verified",
     "sources": [
      "s25",
      "s26",
      "s28",
      "s32",
      "s31"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "Builder. Reports ASIMOV results in its Gemini Robotics reports and model card.",
       "level": "verified",
       "sources": [
        "s25",
        "s26",
        "s28"
       ]
      },
      {
       "value": "Apple",
       "display": "Co-authors of a research paper that reuses ASIMOV-2.0-Constraints (2026-10).",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "Mitsubishi Electric Research Laboratories",
       "display": "Co-authors of VLESA, which evaluates on ASIMOV-2.0-Video (2026-06).",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "ASIMOV-Agentic"
     ],
     "display": "Google DeepMind's 2026 extension to agent safety: refusing unsafe tool calls, stopping near people and asking for help. Separate Atlas entry (asimov-agentic).",
     "level": "verified",
     "sources": [
      "s29",
      "s30"
     ],
     "checked": "2026-10-10",
     "short": "ASIMOV-Agentic (2026, separate entry)"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Most of the advertised data was never released",
     "text": "The CoRL abstract describes 500k situations and 3M instructions. The public bucket holds the validation sets, the Dilemmas-SciFi training set and a 7-item RoboPAIR set. The training sets for Multimodal-Auto (288,421 instructions), Injury (2,335,361) and Dilemmas-Auto (262,621) are not in it. A user asked for the Multimodal-Auto training set on 2025-06-22 (issue #2); there is no reply. The paper also mentions a test split for each component, which is not described or released.",
     "level": "verified",
     "sources": [
      "s3",
      "s1",
      "s19",
      "s17",
      "s18"
     ],
     "status": "open",
     "short": "Most of the 2.9 million v1 instructions are not public."
    },
    {
     "id": "i2",
     "type": "contamination",
     "title": "Answers are public and the builder trains on similar data",
     "text": "All released evaluation sets include their answers, and there is no hidden test set, so later models may have seen them in training. No study has measured this. Google's Gemini Robotics report says Gemini Robotics-ER models are post-trained on ASIMOV-type instances. The best published Constraints score comes from Gemini Robotics-ER 1.5 fine-tuned on 200 pairs made 'using the same synthetic data generation recipe and human annotation process'.",
     "level": "inferred",
     "sources": [
      "s19",
      "s20",
      "s25",
      "s5",
      "s26"
     ],
     "status": "open",
     "note": "The contamination risk is our inference from public answers; the training statements are verified in the cited papers.",
     "short": "The answers are public. Google trains its own models on data made the same way."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "A results figure changed between paper versions and the text was not updated",
     "text": "In the ASIMOV-2.0 paper of 2025-09-25, Figure 5(a) gives video risk accuracy of 0.89 (Gemini 2.5 Pro), 0.64 (Claude Opus 4.1) and 0.52 (GPT-5). The 2025-11-21 version shows 0.84, 0.64 and 0.44. Both versions say the text-to-video gap for GPT-5 is 40%, which matches the old figure (0.92 to 0.52) but not the new one (0.92 to 0.44, a 48-point gap). Figure 10 also changed; the other result figures are byte-identical. No changelog explains the change.",
     "level": "verified",
     "sources": [
      "s7",
      "s5",
      "s8",
      "s9"
     ],
     "status": "open",
     "short": "Video scores changed between two versions of the paper. The text was not updated to match."
    },
    {
     "id": "i4",
     "type": "inconsistent-reporting",
     "title": "Released file counts differ from the paper",
     "text": "v1 Table 1 lists 319 Injury validation instructions; the released injury_val file holds 304 records with one instruction each. Dilemmas-Auto validation is listed as 100 contexts and 200 instructions; the file holds 34 records. Dilemmas-SciFi training is listed as 9,056 contexts; the file holds 9,004. Some of the gaps may come from different counting units, which the release does not document.",
     "level": "verified",
     "sources": [
      "s1",
      "s20",
      "s19",
      "s21"
     ],
     "status": "open",
     "short": "Some public v1 files hold fewer items than the paper lists."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Recognising a danger is different from acting safely",
     "text": "ASIMOV asks models to judge scenarios; it does not test whether they act safely. SafetyALFRED (ACL 2026 Findings) found models that recognised kitchen hazards in question form often failed to deal with them when planning, and concluded that static question-answer evaluations are insufficient for physical safety; it names ASIMOV as such a benchmark but did not use its data. Another 2026 paper describes ASIMOV as testing whether models judge scenarios as safe 'without executing them'.",
     "level": "verified",
     "sources": [
      "s34",
      "s33"
     ],
     "status": "open",
     "short": "A model that recognises danger in a question may still fail to act safely."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Images and videos are AI-generated",
     "text": "v1 images are real frames edited by Imagen 3, and the 2.0 videos and constraint images are generated by Veo 3 and Imagen 3. The 2.0 paper filtered videos that were not photorealistic or broke physics. VLESA notes that robustness on real egocentric video 'remains unverified'. HomeSafe-Bench says ASIMOV-v2 targets general hazards and lacks diversity for household agent behaviours, and a 2026 collision-grounding paper notes it has no human-robot co-presence, depth or physics labels.",
     "level": "verified",
     "sources": [
      "s1",
      "s5",
      "s31",
      "s35",
      "s36"
     ],
     "status": "open",
     "short": "The scenes are generated, so results on real camera footage have not been tested."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Many test items have no human labels",
     "text": "In v1, 1,140 of the 2,273 validation instructions have human labels, and Multimodal-Manual has none (its data was written by people). The paper describes 'a round of human voting' without voter counts or agreement figures. ASIMOV-2.0 used 5 raters per item and dropped low-consensus items, which also removes the hardest or most disputed cases, and reports no agreement statistic.",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s5"
     ],
     "status": "open",
     "note": "That consensus filtering removes disputed cases is our reading.",
     "short": "Half of the v1 test items have no human labels. No figure for agreement between raters is reported."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "ASIMOV measures whether a model's answers about physical danger match human raters. A high score shows the model recognises risks described in text or shown in generated media. It does not show that a robot run by the model will act safely.",
     "basis": [
      "facts.metric_detail",
      "facts.sim_to_real",
      "issues.i5",
      "issues.i6"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "ASIMOV tests how a model judges danger. It does not test whether a robot acts safely."
    },
    {
     "id": "r2",
     "text": "The most useful information is the gap between question types. Frontier models (the most capable current models) score close to the maximum on text risk questions. They score far below it on video timing questions and on questions about robot limits.",
     "basis": [
      "facts.top_score"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Text questions are nearly solved. Video questions and questions about robot limits are not."
    },
    {
     "id": "r3",
     "text": "Most published results come from Google, which also trains its robot models on similar data, and the answers are public. Compare scores across companies with care.",
     "basis": [
      "facts.used_by",
      "issues.i2",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Most published results come from Google. Compare scores across companies with care."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "Whether a robot run by the model acts safely.",
     "sub": "Models answer questions, and no robot moves.",
     "basis": [
      "facts.venue",
      "issues.i5"
     ]
    },
    {
     "id": "l2",
     "text": "How well a model judges danger in real camera footage.",
     "sub": "The images and videos in the test are AI-generated.",
     "basis": [
      "issues.i6"
     ]
    },
    {
     "id": "l3",
     "text": "How a model does on questions it has not seen before.",
     "sub": "The answers are public, and there is no hidden test set.",
     "basis": [
      "issues.i2"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "validity and sim_to_real",
     "where": "v1 arXiv and PMLR camera-ready; ASIMOV-2.0 arXiv v1 and v2 (all figures viewed); project pages v1 and v2; Gemini Robotics (2503.20020), Gemini Robotics 1.5 (2510.03342), Gemini Robotics ER 2 model card and release post, Gemini Robotics 2 safety report; all 43 and 15 Semantic Scholar citing papers scanned by title and 20 opened (VLESA, Visual Grounding Safety, SafetyALFRED, HomeSafe-Bench, Veo world-simulator paper 2512.10675 and others). The scores are themselves agreement with human raters; no study compares ASIMOV scores with robot behaviour, and no rater-agreement statistic is published. Validity list left empty.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code, license_data, license_assets",
     "where": "asimov-benchmark/code tree and licence field; v1 and v2 project pages; dataset_info.json for all 11 released datasets; TFDS catalog pages; TFDS builder source (asimov.py, asimov_v2.py).",
     "date": "2026-10-10"
    },
    {
     "for": "issues.i1 (training sets)",
     "where": "Public bucket listing for prefix robotics/asimov (55 objects, 11 datasets), the official notebook's dataset list, GitHub issue #2.",
     "date": "2026-10-10"
    },
    {
     "for": "ASIMOV-2.0 venue",
     "where": "arXiv record (report number GDM-01-01, no journal reference), v2 project page, web search. No conference or journal found.",
     "date": "2026-10-10"
    },
    {
     "for": "Gemini Robotics 2 / ER 2 ASIMOV-2.0 numbers",
     "where": "ER 2 model card (HTML and PDF), Gemini Robotics 2 release post, Gemini Robotics 2 safety report. Only ASIMOV-Agentic numbers are given; ASIMOV v2 is linked without numbers.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Generating Robot Constitutions & Benchmarks for Semantic Safety (ASIMOV v1, full text)",
     "url": "https://arxiv.org/html/2503.08663v1",
     "type": "paper",
     "publisher": "arXiv (Google DeepMind)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "ASIMOV v1 arXiv abstract page (submission history)",
     "url": "https://arxiv.org/abs/2503.08663",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "ASIMOV v1, CoRL 2025 proceedings page (PMLR 305)",
     "url": "https://proceedings.mlr.press/v305/sermanet25a.html",
     "type": "paper",
     "publisher": "PMLR / CoRL 2025",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "ASIMOV v1, CoRL 2025 camera-ready PDF",
     "url": "https://raw.githubusercontent.com/mlresearch/v305/main/assets/sermanet25a/sermanet25a.pdf",
     "type": "paper",
     "publisher": "PMLR / CoRL 2025",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Can AI Perceive Physical Danger and Intervene? (ASIMOV-2.0, full text v2)",
     "url": "https://arxiv.org/html/2509.21651v2",
     "type": "paper",
     "publisher": "arXiv (Google DeepMind)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "ASIMOV-2.0 arXiv abstract page (v1 2025-09-25, v2 2025-11-21)",
     "url": "https://arxiv.org/abs/2509.21651",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "ASIMOV-2.0 full text, version 1",
     "url": "https://arxiv.org/html/2509.21651v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "ASIMOV-2.0 v1 Figure 5(a): risk recognition, text vs video",
     "url": "https://arxiv.org/html/2509.21651v1/safety/video-risk.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "ASIMOV-2.0 v2 Figure 5(a): risk recognition, text vs video",
     "url": "https://arxiv.org/html/2509.21651v2/safety/video-risk1.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "ASIMOV-2.0 Figure 4(a): risk type and severity accuracy",
     "url": "https://arxiv.org/html/2509.21651v2/safety/injury-latent-risks.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "ASIMOV-2.0 Figure 4(b): action effect and activated risk accuracy",
     "url": "https://arxiv.org/html/2509.21651v2/safety/action-effects-injury.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "ASIMOV-2.0 Figure 5(b): intervention timing",
     "url": "https://arxiv.org/html/2509.21651v2/safety/video-results.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "ASIMOV-2.0 Figure 6(a): constraint violation rates",
     "url": "https://arxiv.org/html/2509.21651v2/safety/constraint-violation-rates.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "ASIMOV Benchmark v1 project page",
     "url": "https://asimov-benchmark.github.io/v1/",
     "type": "site",
     "publisher": "Google DeepMind",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "ASIMOV Benchmark v2 project page",
     "url": "https://asimov-benchmark.github.io/v2/",
     "type": "site",
     "publisher": "Google DeepMind",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "GitHub API: asimov-benchmark/code (stars, forks, licence, last push)",
     "url": "https://api.github.com/repos/asimov-benchmark/code",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "ASIMOV_Datasets.ipynb (official loading notebook and dataset list)",
     "url": "https://github.com/asimov-benchmark/code/blob/main/ASIMOV_Datasets.ipynb",
     "type": "repo",
     "publisher": "ASIMOV authors",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "asimov-benchmark/code issue #2: Multimodal-Auto train release (no reply)",
     "url": "https://github.com/asimov-benchmark/code/issues/2",
     "type": "repo",
     "publisher": "GitHub user",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Google Cloud Storage listing, gresearch bucket, prefix robotics/asimov (object names, sizes, upload dates)",
     "url": "https://storage.googleapis.com/storage/v1/b/gresearch/o?prefix=robotics/asimov",
     "type": "repo",
     "publisher": "Google Research",
     "date": "2025-09-24",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "asimov_injury_val dataset_info.json (record counts; other datasets read at the same path pattern)",
     "url": "https://storage.googleapis.com/gresearch/robotics/asimov_injury_val/0.1.0/dataset_info.json",
     "type": "repo",
     "publisher": "Google Research",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "TensorFlow Datasets catalog: asimov_injury_val",
     "url": "https://www.tensorflow.org/datasets/catalog/asimov_injury_val",
     "type": "repo",
     "publisher": "TensorFlow Datasets (Google)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "TFDS builder source for ASIMOV and ASIMOV v2 datasets",
     "url": "https://github.com/tensorflow/datasets/tree/master/tensorflow_datasets/robotics/asimov",
     "type": "repo",
     "publisher": "TensorFlow Datasets (Google)",
     "date": "2025-09-24",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Semantic Scholar API record for arXiv:2503.08663 (v1)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2503.08663?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Semantic Scholar API record for arXiv:2509.21651 (ASIMOV-2.0)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2509.21651?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Gemini Robotics: Bringing AI into the Physical World (Section 5, Figure 29)",
     "url": "https://arxiv.org/html/2503.20020",
     "type": "paper",
     "publisher": "Google DeepMind",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Gemini Robotics 1.5 tech report (v3; ASIMOV-2.0 section, Figures 18 and 19)",
     "url": "https://arxiv.org/html/2510.03342",
     "type": "paper",
     "publisher": "Google DeepMind",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Gemini Robotics 1.5 Figure 19: ASIMOV-2.0 safety evaluations",
     "url": "https://arxiv.org/html/2510.03342v3/safety_plot_new.png",
     "type": "paper",
     "publisher": "Google DeepMind",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Gemini Robotics ER 2 model card (links semantic safety benchmarks to ASIMOV v2)",
     "url": "https://deepmind.google/models/model-cards/gemini-robotics-er-2/",
     "type": "site",
     "publisher": "Google DeepMind",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Gemini Robotics 2: Safety Evaluations (ASIMOV-Agentic report)",
     "url": "https://storage.googleapis.com/deepmind-media/gemini-robotics/Gemini-Robotics-2-Safety.pdf",
     "type": "paper",
     "publisher": "Google DeepMind",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Gemini Robotics 2 release post (safety section introducing ASIMOV-Agentic)",
     "url": "https://deepmind.google/blog/gemini-robotics-2-brings-whole-body-intelligence-to-robots/",
     "type": "site",
     "publisher": "Google DeepMind",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "VLESA: Vision-Language Embodied Safety Agent for Human Activity Monitoring",
     "url": "https://arxiv.org/html/2606.03954",
     "type": "paper",
     "publisher": "arXiv (Carnegie Mellon University, MERL, Harvard)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Visual Grounding Safety in Vision-Language Models",
     "url": "https://arxiv.org/abs/2610.05637",
     "type": "paper",
     "publisher": "arXiv (UC Riverside, Apple)",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "What the Guard Misses, the Robot Executes: Implied Harm in VLA Instructions",
     "url": "https://arxiv.org/abs/2610.05818",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "SafetyALFRED: Evaluating Safety-Conscious Planning of Multimodal Large Language Models (ACL 2026 Findings)",
     "url": "https://arxiv.org/abs/2604.19638",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "HomeSafe-Bench: Evaluating Vision-Language Models on Unsafe Action Detection for Embodied Agents in Household Scenarios",
     "url": "https://arxiv.org/abs/2603.11975",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Probing Collision Grounding in Vision-Language Models for Safe Human-Robot Collaboration",
     "url": "https://arxiv.org/abs/2605.31196",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "asimov_v2_videos dataset_info.json (287 records)",
     "url": "https://storage.googleapis.com/gresearch/robotics/asimov_v2_videos/0.1.0/dataset_info.json",
     "type": "repo",
     "publisher": "Google Research",
     "date": "2025-09-24",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/frontier-labs.json. New findings: most v1 training data unreleased; ASIMOV-2.0 Figure 5(a) changed between arXiv versions without the text; Google post-trains on ASIMOV-type data; outside uses by VLESA and an Apple co-authored paper. ASIMOV-Agentic kept as a separate entry."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "asimov-agentic",
   "name": "ASIMOV-Agentic",
   "aliases": [
    "Agentic Safety and Uncertainty Resolution Benchmark",
    "google/asimov_agentic"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "An embodied safety benchmark for the planning model that directs a robot (refusals, stops, asking for help). The scope rule includes embodied safety benchmarks.",
   "summary": {
    "text": "Tests whether a robot's planning model refuses unsafe tasks, stops near people and asks for help when unsure.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Google DeepMind (Gemini Robotics Team)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Report 'Gemini Robotics 2: Safety Evaluations', Gemini Robotics Team, Google DeepMind; contributors listed alphabetically. Dataset under the 'google' organisation on Hugging Face."
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2026-07 (HF dataset created 2026-07-21; safety report dated 2026-07-29; announced in GR 2 blog 2026-07-30)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-07 (HF dataset last modified 2026-07-24)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "Initial release (no version label)",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Report says future versions will test longer contexts and 'attention jailbreaking'."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "'offline one-step and multiturn (using VLA tool emulation) benchmark'. A Gemini-based VLA confidence emulator replaces real robots in the multi-turn variant."
    },
    "capability": {
     "value": [
      "safety",
      "embodied-reasoning",
      "collaboration"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Components: unsafe task refusal and safety-constraint following; proactive human-proximity monitoring; safety tool calling on fault messages; VLA feasibility awareness; instruction ambiguity (10 ambiguity classes) and asking for clarification; obfuscated instrument reading."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "tabletop",
      "industrial",
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "ALOHA tabletop scenes, industrial instrument reading (gauges, sight glasses), garage sorting task."
    },
    "scale": {
     "value": "Six components; exact item counts not stated in the report. Hugging Face size category tag: n<1K.",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Files include constraints/physical_constraints.parquet, gauge_reading/*.parquet and ~30 human_safety_monitoring/*.parquet episodes. The dataset card text is behind a Hugging Face login, so per-file counts were not read."
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Self-reported. Frontier models compared: Gemini Robotics ER 2, Claude Opus 4.8, GPT 5.5. Agentic human-monitoring: no model achieves both low FNR and low FPR; FPR under 5% comes with FNR above 40%."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Results only in the safety report figures."
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "HF card metadata license: cc-by-4.0. Access gated ('auto' approval) on Hugging Face."
    },
    "license_code": {
     "value": "CC-BY-4.0 (inferred)",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Evaluation scripts (asimov_agentic_evals.py etc.) ship inside the same HF dataset repo, whose card declares CC-BY-4.0; no separate code licence seen."
    },
    "access": {
     "value": "registration",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Gated with automatic approval; requires a Hugging Face login."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "One real-world check by Google: Apollo 2 humanoid safe stopping (99% detection, 96% safe-pose reliability, lab settings). No correlation statistic between offline scores and real behaviour."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "https://storage.googleapis.com/deepmind-media/gemini-robotics/Gemini-Robotics-2-Safety.pdf",
     "url": "https://storage.googleapis.com/deepmind-media/gemini-robotics/Gemini-Robotics-2-Safety.pdf",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Careers at Google DeepMind — Google DeepMind",
     "url": "https://deepmind.google/careers/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "google/asimov_agentic on Hugging Face (dataset)",
     "url": "https://huggingface.co/api/datasets/google/asimov_agentic",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "autoeval",
   "name": "AutoEval",
   "full_name": "AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World",
   "aliases": [
    "Bridge-AutoEval",
    "AutoEval (Berkeley)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores generalist manipulation policies by closed-loop runs on real WidowX robots, with public job submission; a real-robot evaluation service.",
   "summary": {
    "text": "AutoEval is a UC Berkeley system that evaluates submitted robot policies on real WidowX arms around the clock, using a learned success classifier and a learned reset policy in place of a human operator. Two cells with four tabletop tasks were open to the public; the submission dashboard was offline on 2026-10-10.",
    "short": "AutoEval is a system that tests submitted robot policies (the models that control robots) on real robot arms without a human operator. Its public robot stations, called cells, are now offline.",
    "sources": [
     "s2",
     "s5",
     "s6"
    ]
   },
   "facts": {
    "kind": {
     "value": "arena",
     "level": "inferred",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Users submit a policy server; the organisers' robots run the evaluation and return a report. Classification by the Atlas."
    },
    "kind_secondary": {
     "value": [
      "platform"
     ],
     "display": "Also an open recipe and code for building autonomous evaluation cells for new tasks",
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Appendix H gives a step-by-step guide (about 3 hours of human effort, 5 hours in total per new task)."
    },
    "version": {
     "value": "0.0.1",
     "display": "Package auto_eval 0.0.1. No tags or releases. Paper arXiv v2; the CoRL 2025 camera-ready corrects the daily throughput in its introduction from 500 to 850 episodes.",
     "level": "verified",
     "sources": [
      "s8",
      "s10",
      "s1",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "arXiv v2's introduction says 500 episodes per 24 hours while its Section 5.3 says about 850; the camera-ready says 850 in both.",
     "short": "0.0.1. There are no tagged releases."
    },
    "publishers": {
     "value": [
      "UC Berkeley",
      "NVIDIA"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Zhiyuan Zhou, Pranav Atreya, You Liang Tan (also NVIDIA), Karl Pertsch, Sergey Levine."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Sergey Levine's lab at UC Berkeley."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Robots at UC Berkeley."
    },
    "first_release": {
     "value": "2025-03",
     "display": "arXiv v1 on 2025-03-31. Published at CoRL 2025 (PMLR volume 305, pages 1997-2017, 2025-10-07). Logged public jobs start 2025-03-12.",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Code repository created 2025-03-26.",
     "short": "March 2025, at CoRL 2025"
    },
    "latest_update": {
     "value": "2026-06",
     "display": "Last logged evaluation job 2026-06-13 (dataset updated 2026-06-14). Last code commit 2026-03-26.",
     "level": "verified",
     "sources": [
      "s11",
      "s12",
      "s10"
     ],
     "checked": "2026-10-10",
     "short": "June 2026, when the last job was logged"
    },
    "status": {
     "value": "maintained",
     "display": "Code fixes until March 2026; the public dashboard was offline on 2026-10-10. The site says the four public tasks run until 2026-01-01 and that 2026 tasks are to be decided.",
     "level": "inferred",
     "sources": [
      "s10",
      "s6",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Jobs were still logged in January (233) and March 2026 (116), then 2 in May and 1 in June 2026. See facts.access.",
     "short": "The last code fix was in March 2026. The public service is offline."
    },
    "capability": {
     "value": [
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Pick-and-place, articulated (drawer) and deformable (cloth folding) tasks."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper randomises the start position of the eggplant, drawer and cloth in each episode; lighting is held constant."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "WidowX 250 6-DoF arm with a Logitech C920 camera (256 x 256 top-down image)",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "End-effector delta actions with blocking control. Policies get the image, an 8-number robot state and the instruction (site)."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Drawer, toy-sink and cloth scenes in the style of the BridgeData V2 setups."
    },
    "tasks": {
     "value": 5,
     "display": "5 tasks in the paper (open drawer, close drawer, eggplant to basket, eggplant to sink, fold cloth) in 3 cells. 4 public tasks in 2 cells (no cloth).",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "5 tasks (4 public) in 3 cells"
    },
    "demonstrations": {
     "value": "none for evaluated policies",
     "display": "No training data for the policies under test. Each cell's reset policy was trained on 50-100 teleoperated demonstrations and its success classifier on about 1,000 labelled images.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Reset policies fine-tune OpenVLA with LoRA; the public cells use a scripted reset for the drawer and a fine-tuned MiniVLA for the sink (Appendix I)."
    },
    "scoring": {
     "value": [
      "success-rate",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Binary success decided by a fine-tuned PaliGemma 3B classifier, deployed only above 95% accuracy. The report gives overall_success_rate. The paper lists binary-only scoring as a limitation."
    },
    "metric_detail": {
     "value": "binary success rate judged by a learned classifier",
     "display": "Success rate over the job's episodes, judged automatically; results come back as a Weights & Biases report with videos, and episodes are logged to Hugging Face.",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "Success rate, judged automatically"
    },
    "trials": {
     "value": "50 per policy and task (paper); public jobs default to 10",
     "level": "verified",
     "sources": [
      "s2",
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper",
       "display": "50 rollouts per policy and task; maximum 70 steps (drawer), 100 (sink), 80 (cloth).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Public web UI",
       "display": "Number of episodes defaults to 10 (allowed 1-50); steps per episode default 70. The server-side default is 50.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      },
      {
       "value": "World-Gymnast",
       "display": "10 trials per policy and task, repeated 5 times to estimate standard error.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      }
     ],
     "short": "50 in the paper and 10 by default"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "The paper's tables give success counts out of 50 without intervals; its consistency figure shows 95% intervals. World-Gymnast reports standard errors from AutoEval runs."
    },
    "evaluator": {
     "value": "organiser-run",
     "level": "verified",
     "sources": [
      "s5",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Users submit a policy-server address; the system queues jobs and runs them on its robots."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s5",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Per-job Weights & Biases reports and a public log dataset; no ranking page."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file, copyright 2025 zhouzypaul; README badge agrees."
    },
    "license_data": {
     "value": "MIT",
     "display": "MIT for the evaluation logs on Hugging Face (dataset card)",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10"
    },
    "access": {
     "value": "closed",
     "display": "The public submission dashboard was offline on 2026-10-10 (ngrok: endpoint offline). Code, logs and the set-up guide remain open (MIT).",
     "level": "inferred",
     "sources": [
      "s6",
      "s5",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The site and README still point to the dashboard. Level is our reading: the service accepts no jobs now, while anyone can build their own cell from the code.",
     "short": "The service is offline. The code is open."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s9",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Code and logs are MIT. Building a cell also involves fine-tuning PaliGemma and OpenVLA, whose own terms were not checked here. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "display": "Scores come from real robots. Agreement of the automated scores with human-run scores is under validity.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "In the same paper, SIMPLER simulations of the same tasks agreed less well with the human-run results (mean r 0.548 by our computation; see the SimplerEnv entry)."
    },
    "real_reproducibility": {
     "value": "protocol-only",
     "display": "One site. Repeat runs at that site were stable; no second site has been set up that we found.",
     "level": "inferred",
     "sources": [
      "s2",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Two policies on three tasks run two months apart changed by -6 to +2 points (Table 5; one cell is mis-computed: 48 to 46 of 50 is -4 points, shown as -2%). Nine back-to-back runs of Open-pi0 stayed within ±10% for the first 350 episodes, then drifted as motors overheated. VLA-REPLICA (2026) rates AutoEval's reproducibility as low."
    },
    "citations": {
     "value": 63,
     "display": "63 (Semantic Scholar; 5 influential)",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "63"
    },
    "github_stars": {
     "value": 108,
     "display": "108 stars, 8 forks (zhouzypaul/auto_eval)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "108"
    },
    "dataset_downloads": {
     "value": 25153,
     "display": "25,153 (Hub 'downloads' field), 49,687 all time: evaluation logs",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API; time window not checked.",
     "short": "25,153 on Hugging Face"
    },
    "used_by": {
     "value": "The public cells logged 1,096 evaluation jobs from 2025-03-12 to 2026-06-13. World-Gymnast (2026-02) used AutoEval as its real-robot test.",
     "level": "verified",
     "sources": [
      "s12",
      "s11",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "Our count from the public job index: 79 jobs in March 2025, 238 in April, 119 in May, 98 in June, 119 in July, 72 in August, 9 in September, 4 in December 2025; 233 in January 2026, 6 in February, 116 in March, 2 in May, 1 in June. All 1,096 index rows name the robot 'widowx_drawer', so the index may not separate the two cells.",
     "items": [
      {
       "value": "World-Gymnast",
       "display": "2026-02: real-robot results on the 4 public tasks for policies trained with RL in a world model and in SIMPLER.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      }
     ],
     "short": "1,096 logged jobs (March 2025 to June 2026)"
    },
    "industry_use": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No company report using AutoEval found in web searches, the citing papers we opened or the logs (which do not name users). NVIDIA appears only as an author affiliation."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "The public service is offline",
     "text": "On 2026-10-10 the dashboard address returned 'The endpoint auto-eval.ngrok.app is offline' (ERR_NGROK_3200). The site says the four public tasks are available until 2026-01-01 and that 2026 tasks are to be decided. The public logs show jobs until 2026-06-13.",
     "level": "verified",
     "sources": [
      "s6",
      "s5",
      "s12"
     ],
     "status": "open",
     "short": "The public dashboard was offline on 2026-10-10."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "The public tests cover few tasks at one site",
     "text": "The public offering was 4 tabletop tasks in 2 cells at Berkeley. The authors list as limits: new scenes need hours of set-up, robustness factors such as camera angle or lighting cannot be varied in a controlled way, mobile manipulation is not covered, and success is binary. VLA-REPLICA (2026) rates AutoEval's task diversity as low (4 tasks) and its reproducibility as low.",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s16"
     ],
     "status": "open",
     "short": "The public service offered four tabletop tasks at one lab."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Automated judging makes errors, most often on cloth folding",
     "text": "Agreement with human judges was near perfect on the drawer and eggplant tasks but lower on cloth folding (Pearson r 0.720 by our computation). Example: Open-pi0 folded the cloth 12/50 times according to AutoEval and 3/50 according to humans. In a 50-trial failure analysis on the sink, 3 trials were wrongly counted as successes because the reset policy failed. The authors suggest humans re-check the report videos when accuracy matters.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "On the cloth-folding task, AutoEval counted 12 successes for one policy where humans counted 3."
    },
    {
     "id": "i4",
     "type": "protocol-variance",
     "title": "Scores drift over long runs and default jobs are small",
     "text": "Scores drifted after about 8 hours of continuous running because the WidowX motors overheat; the cells pause for 20 minutes every 6 hours. The public web UI defaults to 10 episodes per job (maximum 50), while the paper used 50.",
     "level": "verified",
     "sources": [
      "s2",
      "s14"
     ],
     "status": "open",
     "short": "Scores drift after about 8 hours of continuous running. Public jobs run 10 episodes by default."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "AutoEval's automation is well validated for its own cells. Automated scores closely matched human-run scores on 4 of 5 tasks. That says little about how a policy does elsewhere, because every test ran on four tabletop tasks in one lab.",
     "basis": [
      "facts.tasks",
      "facts.real_reproducibility",
      "issues.i3"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Automated judging matches human judges in these cells. The tasks cover a narrow range."
    },
    {
     "id": "r2",
     "text": "With a 10-episode job, the 95% confidence interval for a success rate near 50% spans about ±30 points (by binomial arithmetic). Use 50 episodes, as the paper did, before you compare policies.",
     "basis": [
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Single 10-episode jobs are too small to compare policies."
    },
    {
     "id": "r3",
     "text": "The dashboard is offline and there is no second site. AutoEval is now mainly an open method for labs that want to automate their own real-robot tests. It includes a published way to check the automation against human judges.",
     "basis": [
      "facts.access",
      "facts.kind_secondary",
      "issues.i1"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "AutoEval is now mostly a method that labs can use to automate their own tests."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How a policy would do in other scenes or labs.",
     "sub": "The tests cover four tabletop tasks at one site.",
     "basis": [
      "facts.tasks",
      "facts.real_reproducibility"
     ]
    },
    {
     "id": "l2",
     "text": "How much partial progress a policy makes, or how well it moves.",
     "sub": "A learned classifier (a model trained to judge the outcome) marks each attempt as a success or a failure.",
     "basis": [
      "facts.scoring"
     ]
    },
    {
     "id": "l3",
     "text": "Whether a small difference in a single job is real.",
     "sub": "Public jobs run 10 episodes (attempts) by default.",
     "basis": [
      "facts.trials"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "AutoEval paper: automated compared with human-run evaluation",
     "date": "2025-03",
     "by": "authors",
     "method": "AutoEval and human operators evaluated the same 6 policies on the same real cells. The policies were OpenVLA, Octo, Open-pi0, MiniVLA, SuSIE and SuSIE's low-level policy, so there were 5 distinct models. Each policy ran 50 trials on each of 5 tasks.",
     "result": "Pearson correlation r = 0.942 and MMRV (a measure of how often two rankings disagree) = 0.015, both as means over 5 tasks",
     "authors_view": "closely match",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Both sides are real-robot runs; this checks the automation, not a simulator. Per task (our computation from Tables 1-2, which reproduces the paper's means): open drawer r 0.994, close drawer 0.997, eggplant to basket 1.000, eggplant to sink 1.000, fold cloth 0.720 (MMRV 0.063). No independent check of AutoEval's judging found."
    }
   ],
   "searched": [
    {
     "for": "validity (independent checks of AutoEval against human-run evaluation)",
     "where": "AutoEval arXiv v2 and camera-ready; World-Gymnast 2602.02454 (uses AutoEval, no human comparison); VLA-REPLICA 2605.20774 (characterisation only); WorldEval 2505.19017, Scalable Policy Evaluation with Video World Models 2511.11520, RoboWorld 2607.01060, RobotArena Infinity 2510.23571, PolaRiS 2512.16881 (cite only); extended web search.",
     "date": "2026-10-10"
    },
    {
     "for": "access",
     "where": "Dashboard URL from the site and README (offline), project site, README, public logs.",
     "date": "2026-10-10"
    },
    {
     "for": "industry_use",
     "where": "Web search, citing papers opened above, public log index.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "No single headline score: per-task counts only. Skipped.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "AutoEval (arXiv abstract page; v1 2025-03-31, v2 2025-04-02)",
     "url": "https://arxiv.org/abs/2503.24278",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "AutoEval paper, full text v2 (Sections 3-5, Fig. 7, Appendices A-L, Tables 1-5)",
     "url": "https://arxiv.org/html/2503.24278v2",
     "type": "paper",
     "publisher": "arXiv (UC Berkeley, NVIDIA)",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "AutoEval, PMLR proceedings page (CoRL 2025, PMLR 305:1997-2017)",
     "url": "https://proceedings.mlr.press/v305/zhou25a.html",
     "type": "paper",
     "publisher": "PMLR (9th Conference on Robot Learning)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "AutoEval camera-ready PDF in PMLR",
     "url": "https://raw.githubusercontent.com/mlresearch/v305/main/assets/zhou25a/zhou25a.pdf",
     "type": "paper",
     "publisher": "PMLR (9th Conference on Robot Learning)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "AutoEval project site (tasks, submission instructions, FAQ)",
     "url": "https://auto-eval.github.io/",
     "type": "site",
     "publisher": "AutoEval team (UC Berkeley)",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "AutoEval dashboard URL (returned 'endpoint offline', ERR_NGROK_3200, on 2026-10-10)",
     "url": "https://auto-eval.ngrok.app/page",
     "type": "site",
     "publisher": "AutoEval team",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: zhouzypaul/auto_eval (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/zhouzypaul/auto_eval",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "AutoEval README",
     "url": "https://github.com/zhouzypaul/auto_eval/blob/main/README.md",
     "type": "repo",
     "publisher": "Zhiyuan Zhou",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "AutoEval LICENSE (MIT)",
     "url": "https://github.com/zhouzypaul/auto_eval/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Zhiyuan Zhou",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "AutoEval commit history, tags and releases",
     "url": "https://github.com/zhouzypaul/auto_eval/commits/main",
     "type": "repo",
     "publisher": "Zhiyuan Zhou",
     "date": "2026-03-26",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Hugging Face dataset zhouzypaul/auto_eval: card (license: mit), API record, 1,096 eval_data job folders",
     "url": "https://huggingface.co/datasets/zhouzypaul/auto_eval",
     "type": "repo",
     "publisher": "Zhiyuan Zhou",
     "date": "2026-06-14",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Hugging Face dataset viewer rows for zhouzypaul/auto_eval (job index: time, robot, location)",
     "url": "https://datasets-server.huggingface.co/rows?dataset=zhouzypaul/auto_eval&config=default&split=test&offset=0&length=100",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Semantic Scholar API record for arXiv:2503.24278",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2503.24278?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "AutoEval web UI source (static/index.html: episodes default 10, max 50; 70 steps) and job_scheduler.py",
     "url": "https://github.com/zhouzypaul/auto_eval/blob/main/static/index.html",
     "type": "repo",
     "publisher": "Zhiyuan Zhou",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "World-Gymnast: Training Robots with Reinforcement Learning in a World Model (Section 4, Appendix D.2)",
     "url": "https://arxiv.org/abs/2602.02454",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "VLA-REPLICA: A Low-Cost, Reproducible Benchmark for Real-World Evaluation of VLA Models (Table 1)",
     "url": "https://arxiv.org/abs/2605.20774",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and the real-eval inventory record. Added per-task agreement, camera-ready corrections, public-UI defaults, job-log counts and the World-Gymnast use."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "behavior-challenge",
   "name": "BEHAVIOR Challenge",
   "aliases": [
    "2025 BEHAVIOR Challenge",
    "2026 BEHAVIOR Challenge",
    "BEHAVIOR Challenge @ NeurIPS 2025",
    "BEHAVIOR Challenge @ CoRL 2026"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Time-bound competition that ranks embodied policies controlling a simulated bimanual mobile robot on household tasks.",
   "summary": {
    "text": "Annual simulation competition on 50 (2025) or 100 (2026) BEHAVIOR-1K household tasks with a bimanual mobile robot.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Stanford Vision and Learning Lab (page footer); 2025 sponsors: Simovation, IMDA, Stanford HAI, Schmidt Family Foundation, NVIDIA; 2026 sponsors: Simovation, IMDA, Stanford HAI, Schmidt Family Foundation",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "2026 sponsors from https://behavior.stanford.edu/challenge/index.html"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-09 (2025 challenge launched 2025-09-02; the 2026 page calls 2026 the 'second year')",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "A predecessor 'BEHAVIOR Challenge 2021' on BEHAVIOR-100 (100 activities, iGibson) ran on EvalAI 2021-07-15 to 2022-04-01: https://eval.ai/web/challenges/challenge-page/1190/overview"
    },
    "latest_update": {
     "value": "2026-10-09: final deadline 2026-10-16 AoE; new rules on evaluation speed, timeouts, reconnections, single-run submissions. 2026-10-07: move to v3.9.3-post2.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "2026 BEHAVIOR Challenge, 100 tasks, single track (RGB + depth + proprioception)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "long-horizon",
      "mobile-manipulation",
      "bimanual",
      "manipulation",
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Bimanual inferred from the R1 Pro robot model (two 7-DOF arms)."
    },
    "embodiment": {
     "value": [
      "mobile-manipulator",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Challenge use from https://behavior.stanford.edu/challenge/archive/2025/evaluation.html and https://behavior.stanford.edu/challenge/evaluation.html. The official pages do not name the manufacturer."
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "trials": {
     "value": "2025: participants self-evaluate on 10 instances per task, 1 rollout each; organisers evaluate top 5 on 10 held-out instances per task. 2026: 20 public + 20 hidden instances per task; reported results on instances 0-9, 1 rollout each; top-5 re-run on hidden instances.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "2025 from https://behavior.stanford.edu/challenge/archive/2025/evaluation.html. The 2026 page describes the hidden set in two slightly different ways."
    },
    "evaluator": {
     "value": "both",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Both (public validation self-reported; held-out test run by organisers)"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Official (: 2025 provisional board on behavior.stanford.edu; 2026 board as a Hugging Face Space (running, updated 2026-10-10)",
     "note": "2026: https://huggingface.co/spaces/behavior-1k/2026-challenge-leaderboard"
    },
    "industry_use": {
     "value": "Industry teams on the 2025 board: NVIDIA Research (2nd), Beijing Simple AI Technology (3rd), Huawei CRI EAI Team (4th), Cloud Data Technology",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "license_code": {
     "value": "MIT (BEHAVIOR-1K repository LICENSE)",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Demos: MIT (HF cards behavior-1k/2025-challenge-demos, behavior-1k/2026-challenge-demos). Assets: BEHAVIOR Data Bundle EULA, non-commercial academic research only.",
     "level": "verified",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Asset EULA from https://github.com/StanfordVL/BEHAVIOR-1K/blob/main/setup.sh; demos card also https://huggingface.co/datasets/behavior-1k/2025-challenge-demos"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No real-robot evaluation of challenge entries found on the challenge pages, the HAI announcement, or the top-2 team reports. The only BEHAVIOR sim-to-real data is the one-task study in the BEHAVIOR-1K paper (see that record)."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "2026 edition open; updates posted 2026-10-09."
    },
    "kind": {
     "value": "challenge",
     "level": "inferred",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "🏆 2026 BEHAVIOR Challenge - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/index.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "🏆 2025 BEHAVIOR Challenge - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/archive/2025/index.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Announcements / Updates - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/updates.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Evaluation and Rules - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/evaluation.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Robots - BEHAVIOR",
     "url": "https://behavior.stanford.edu/omnigibson/robots.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "Removing Bottlenecks: Training Generalist Robotics Policies for the BEHAVIOR-1K Challenge | Papers with Code",
     "url": "https://paperswithcode.co/paper/104247",
     "type": "secondary",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "Challenge Leaderboard - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Baselines - BEHAVIOR",
     "url": "https://behavior.stanford.edu/challenge/baselines.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "StanfordVL/BEHAVIOR-1K on GitHub (blob)",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "behavior-1k/2026-challenge-demos on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/behavior-1k/2026-challenge-demos",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s11": {
     "title": "BEHAVIOR-1K: A Human-Centered, Embodied AI Benchmark with 1,000 Everyday Activities and Realistic Simulation",
     "url": "https://arxiv.org/abs/2403.09227",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-03"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "behavior-1k",
   "name": "BEHAVIOR-1K",
   "full_name": "BEHAVIOR-1K: A Human-Centered, Embodied AI Benchmark with 1,000 Everyday Activities and Realistic Simulation",
   "aliases": [
    "BEHAVIOR",
    "B1K",
    "BEHAVIOR-1K / OmniGibson",
    "BEHAVIOR Challenge",
    "2025 BEHAVIOR Challenge",
    "2026 BEHAVIOR Challenge"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "BEHAVIOR-1K is a Stanford simulation benchmark of 1,000 household activities, defined in a logic language and run in the OmniGibson simulator on NVIDIA Isaac Sim. Its yearly BEHAVIOR Challenge (50 tasks in 2025, 100 in 2026) scores a two-armed mobile robot by the share of goal conditions it meets.",
    "sources": [
     "s1",
     "s5",
     "s14",
     "s20"
    ],
    "short": "BEHAVIOR-1K is a set of 1,000 household activities in simulation. Its yearly challenge scores a two-armed mobile robot on a subset of these activities."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper calls BEHAVIOR-1K a simulation benchmark with fixed activity definitions (BDDL) and metrics."
    },
    "kind_secondary": {
     "value": [
      "platform",
      "challenge"
     ],
     "display": "Also a simulator (OmniGibson) and a yearly competition (the BEHAVIOR Challenge, 2025 and 2026). Stanford HAI announced the 2025 edition on 2025-09-22 with a top prize of $1,000.",
     "level": "verified",
     "sources": [
      "s2",
      "s14",
      "s20",
      "s43"
     ],
     "checked": "2026-10-10",
     "note": "This record covers the benchmark and both challenge editions. A separate basic record (behavior-challenge) also exists."
    },
    "publishers": {
     "value": [
      "Stanford Vision and Learning Lab"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "2024 paper: 35 authors. Funders listed include Stanford HAI, Toyota Research Institute, NSF, ONR, Amazon, Bosch, Salesforce and Samsung.",
     "items": [
      {
       "value": "Stanford University (Stanford Vision and Learning Lab)",
       "display": "Lead organisation. Site footer: '© 2026 Stanford Vision and Learning Lab'. Most authors are at Stanford.",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ]
      },
      {
       "value": "Co-author affiliations (2024 paper)",
       "display": "The University of Texas at Austin, University of Illinois Urbana-Champaign, University of Southern California, Salesforce Research.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "NVIDIA Research (internships)",
       "display": "Acknowledgements: work done in part while three authors were interns at Nvidia Research.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Challenge sponsors",
       "display": "2026 page logos: Simovation, IMDA, Stanford HAI, Schmidt Family Foundation. NVIDIA joined as a 2025 sponsor (announcement of 2025-10-08).",
       "level": "verified",
       "sources": [
        "s14",
        "s20"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Lead organisation is a university lab."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Stanford University, California, USA."
    },
    "first_release": {
     "value": "2022-12",
     "display": "CoRL 2022 (Auckland, 14 to 18 December 2022), PMLR volume 205. Extended version on arXiv on 2024-03-14.",
     "level": "verified",
     "sources": [
      "s3",
      "s1",
      "s7",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The CoRL title is 'BEHAVIOR-1K: A Benchmark for Embodied AI with 1,000 Everyday Activities and Realistic Simulation'. The GitHub repository was created on 2021-12-17. The first release tag, OmniGibson v0.0.1, is dated 2022-12-16. The 2025 challenge launched on 2025-09-02.",
     "short": "December 2022, at CoRL 2022"
    },
    "latest_update": {
     "value": "2026-10",
     "display": "Release v3.9.3-post2 on 2026-10-07 (one fix for custom-robot evaluation). New challenge rules posted 2026-10-09.",
     "level": "verified",
     "sources": [
      "s6",
      "s16",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Five releases since 2026-07-03. Repository last pushed 2026-10-10 (GitHub API).",
     "short": "October 2026 (v3.9.3-post2)"
    },
    "version": {
     "value": "v3.9.3-post2",
     "display": "BEHAVIOR-1K v3.9.3-post2. The 2025 challenge ran on the v3.7 series with Isaac Sim 4.5.0; the 2026 challenge runs on the v3.9 series with Isaac Sim 5.1.0.",
     "level": "verified",
     "sources": [
      "s6",
      "s9",
      "s10",
      "s15",
      "s21"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v3.7.0 to v3.7.2",
       "display": "2025-09-02 to 2025-12-15. Used for the 2025 challenge. setup.sh installs Isaac Sim 4.5.0.",
       "level": "verified",
       "sources": [
        "s6",
        "s10"
       ]
      },
      {
       "value": "v3.9.0 to v3.9.3-post2",
       "display": "2026-07-03 to 2026-10-07. Used for the 2026 challenge. setup.sh installs Isaac Sim 5.1.0.",
       "level": "verified",
       "sources": [
        "s6",
        "s9"
       ]
      },
      {
       "value": "OmniGibson v0.0.1 to v1.1.1",
       "display": "Earlier tags of the same repository, 2022-12-16 to 2024-10-04, released under the OmniGibson name.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "2025 BEHAVIOR Challenge",
       "display": "50 tasks. Two tracks: standard (onboard RGB, depth, segmentation and proprioception) and privileged information. Robot fixed to R1 Pro.",
       "level": "verified",
       "sources": [
        "s20",
        "s21"
       ]
      },
      {
       "value": "2026 BEHAVIOR Challenge",
       "display": "100 tasks in 7 scenes (4 new). One track (RGB, depth and proprioception). Robot not fixed: R1 Pro by default or another OmniGibson robot.",
       "level": "verified",
       "sources": [
        "s14",
        "s15"
       ]
      }
     ],
     "short": "v3.9.3-post2, released October 2026"
    },
    "capability": {
     "value": [
      "long-horizon",
      "mobile-manipulation",
      "bimanual",
      "manipulation",
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "The paper calls the activities long-horizon and dependent on complex manipulation. The 2026 page says BEHAVIOR tests reasoning, long-horizon navigation and dexterous bimanual manipulation in house-scale scenes. Putting long-horizon first is our reading. Bimanual applies to the challenge robot."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "Challenge test instances keep the same tasks as the training demonstrations. Only the starting states of task objects and the robot's starting pose change.",
     "level": "verified",
     "sources": [
      "s15",
      "s21",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "2026 rules: 'Each instance differs in terms of initial object states and initial robot poses.' The first-place team calls the required generalisation 'very limited' and notes there is no test of unseen object categories, language goals or new tasks. No official train and test split exists for the full 1,000 activities.",
     "short": "Only the start positions change"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "OmniGibson on NVIDIA Isaac Sim 5.1.0",
     "display": "OmniGibson, built on NVIDIA Omniverse and PhysX 5, installed with Isaac Sim 5.1.0 (v3.9 series) or 4.5.0 (v3.7 series). Simulates rigid and deformable bodies, cloth, fluids and object states such as temperature.",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s10",
      "s11",
      "s42"
     ],
     "checked": "2026-10-10",
     "note": "2024 paper: about 60 fps for a house scene with about 60 objects, ray-traced. Requirements page: Ubuntu 22.04+ or Windows 10+, 32 GB RAM, NVIDIA RTX 2070 or better with 8 GB VRAM. NVIDIA's documentation page for Isaac Sim 5.1.0 carries a banner saying that release is no longer supported (checked 2026-10-10).",
     "short": "OmniGibson on Isaac Sim 5.1"
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "display": "Not fixed by the benchmark. The challenge default is the R1 Pro, a two-armed robot on a wheeled base. The 2026 rules also allow other OmniGibson robots.",
     "level": "verified",
     "sources": [
      "s13",
      "s15",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "OmniGibson supports 12 robots: 4 mobile robots, 3 manipulators, 4 mobile manipulators and a bimanual proxy for VR teleoperation. The taxonomy value 'bimanual-arm' means two arms on a fixed base, so it is not used for the R1 Pro."
    },
    "robots": {
     "value": "R1 Pro (simulated), challenge default",
     "display": "R1 Pro: holonomic base, 4-DOF torso, two 7-DOF arms and two parallel-jaw grippers. The 2022 real-robot study used a PAL Robotics Tiago++.",
     "level": "verified",
     "sources": [
      "s13",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "R1 Pro maker",
       "display": "Not named on the BEHAVIOR pages. The same team's BRS paper names the smaller R1 as a Galaxea robot, and Galaxea's G0.5 report fine-tunes on real R1-Pro robots. So the maker is most likely Galaxea.",
       "level": "inferred",
       "sources": [
        "s44",
        "s35"
       ]
      }
     ],
     "short": "R1 Pro (simulated)"
    },
    "scene": {
     "value": [
      "home",
      "office-lab",
      "mixed"
     ],
     "display": "50 scenes: houses, gardens, restaurants, offices, grocery stores and others. The 2026 challenge uses 7 scenes.",
     "level": "verified",
     "sources": [
      "s1",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Gardens, restaurants and stores have no exact taxonomy value, so 'mixed' is added."
    },
    "tasks": {
     "value": 1000,
     "display": "1,000 activities defined in BDDL. The challenge uses 50 of them (2025) and 100 (2026).",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s14",
      "s20",
      "s16",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "The 1,000 are the 909 activities ranked highest in a survey of 1,461 people plus 91 activities from BEHAVIOR-100 (2024 paper). The 2026 set keeps the 50 tasks of 2025 and adds 50.",
     "short": "1,000 activities, with 50 or 100 of them in the challenge"
    },
    "scenes": {
     "value": 50,
     "display": "50 interactive scenes with 373 rooms: 15 houses from BEHAVIOR-100 and 35 new scenes",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "50 scenes"
    },
    "objects": {
     "value": 9318,
     "display": "9,318 object models in 1,949 categories (2024 paper, Table 1). The CoRL 2022 abstract says more than 5,000 objects; the site now says 10,000+.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Three primary sources give three numbers as the asset set grew. The 2024 text rounds to '9,000+ object models from 1,900+ categories'.",
     "short": "9,318 object models (2024 paper)"
    },
    "demonstrations": {
     "value": 20000,
     "display": "20,000 teleoperated demonstrations for the 2026 challenge (1,950 hours, 200 per task). The 2025 challenge had 10,000.",
     "level": "verified",
     "sources": [
      "s14",
      "s18",
      "s25",
      "s20",
      "s24",
      "s15"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2026: 20,000 demos, 1,950 hours",
       "display": "100 tasks. 210,916,774 frames. Average trajectory 351.54 s. 3.27 TB in LeRobot v3 format; raw HDF5 1.44 TB. Collected with the JoyLo whole-body teleoperation interface; data provided by Simovation.",
       "level": "verified",
       "sources": [
        "s14",
        "s18",
        "s25"
       ]
      },
      {
       "value": "2025: 10,000 demos, '1200+ hours'",
       "display": "50 tasks, 200 per task. The dataset card lists 119,094,660 frames at 30 fps.",
       "level": "verified",
       "sources": [
        "s20",
        "s24"
       ]
      },
      {
       "value": "2025 hours: conflict",
       "display": "119,094,660 frames at 30 fps is about 1,103 hours by our arithmetic. The challenge page says 1200+ hours; Galaxea's G0.5 report says over 1,100 hours.",
       "level": "inferred",
       "sources": [
        "s24",
        "s20",
        "s35"
       ],
       "note": "CONFLICT between the challenge page and the dataset card."
      },
      {
       "value": "Dataset corrections in 2026",
       "display": "Base velocity frame fixed (2026-07-27), depth videos fixed (2026-07-27), arm, gripper and trunk velocity fields fixed (2026-08-24).",
       "level": "verified",
       "sources": [
        "s16"
       ]
      }
     ],
     "short": "20,000 teleoperated (remote-controlled) demonstrations for 2026"
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s15",
      "s21",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "Ranking uses the partial-credit Q-score. Full success rate is also reported."
    },
    "metric_detail": {
     "value": "Q-score (share of BDDL goal conditions met)",
     "display": "Task success score Q: the share of BDDL goal conditions satisfied at the end of an episode, using the best-matched goal clause, averaged over instances and tasks. Full success rate is reported too. Ties are broken by simulated time, base distance and hand movement, normalised by human averages from 200 demonstrations per task.",
     "level": "verified",
     "sources": [
      "s15",
      "s22",
      "s2",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "The 2022 and 2024 papers report success rate, Q and three efficiency metrics. In the first-place entry, partial successes make up roughly half of the score (team report).",
     "short": "Share of goal conditions met (Q-score)"
    },
    "trials": {
     "value": "1 rollout on each of 10 instances per task",
     "level": "verified",
     "sources": [
      "s21",
      "s22",
      "s15",
      "s16",
      "s17",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2025 challenge",
       "display": "Teams ran 1 rollout on each of 10 public instances per task (500 rollouts) and reported the scores. Timeout: 2 times the average human completion time. The organisers re-ran the top 5 on 10 held-out instances per task.",
       "level": "verified",
       "sources": [
        "s21",
        "s22"
       ]
      },
      {
       "value": "2026 challenge",
       "display": "100 tasks x 10 instances x 1 rollout = 1,000 rollouts. Timeout: 1.5 times the mean human demonstration length. From 2026-10-09 each rollout must also average at least 1 FPS. The organisers evaluate top submissions on hidden instances.",
       "level": "verified",
       "sources": [
        "s15",
        "s17",
        "s16"
       ],
       "note": "The rules describe the hidden set as indices 20 to 39 in one place and as 10 more held-out instances in another."
      },
      {
       "value": "Nondeterminism",
       "display": "The organisers state the simulator is nondeterministic, so rollouts of the same policy on one instance can differ. They forbid picking the best of repeated runs.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      },
      {
       "value": "2024 paper baselines",
       "display": "Trained with seeds 0, 1 and 2 and evaluated with seed 0.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "1 run on 10 instances per task"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s22",
      "s31",
      "s35"
     ],
     "checked": "2026-10-10",
     "note": "The original papers report means with a spread over training seeds. The challenge leaderboards and the team reports we read give single numbers. G0.5 averages two evaluation runs without an interval."
    },
    "evaluator": {
     "value": "both",
     "display": "Scores on public instances are self-reported. Held-out scores for the top teams are run by the organisers.",
     "level": "verified",
     "sources": [
      "s22",
      "s23",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "2025 leaderboard: 'Public Validation' entries are self-reported; 'Held-out Test' entries are verified by the BEHAVIOR team. 2026: final evaluation by the organisers from a Docker image or a policy server reached over the internet.",
     "short": "Teams report their own scores. The organisers re-run the top entries."
    },
    "leaderboard": {
     "value": "official",
     "display": "Official boards for the challenge task sets only: 2025 (provisional, on behavior.stanford.edu) and 2026 (Hugging Face Space). No board covers the full 1,000 activities.",
     "level": "verified",
     "sources": [
      "s22",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "2026: self-reported scores were shown from 2026-08-31 (commit 'Show self-reported leaderboard scores') and withheld on 2026-10-09 pending verification, as announced on the updates page. Winners are due 2026-11-04. We do not reproduce the withheld 2026 scores here.",
     "short": "Official, for the challenge task sets only"
    },
    "top_score": {
     "value": 0.2599,
     "display": "Best organiser-verified result: Q 0.2599 on the 2025 held-out test (Robot Learning Collective). Later self-reported results on the 2025 task set reach 0.3453 (Comet, after the challenge) and 0.3136 (Galaxea G0.5). Rows use different protocols and are not directly comparable. The chart shows Q x 100.",
     "level": "inferred",
     "sources": [
      "s22",
      "s32",
      "s35"
     ],
     "checked": "2026-10-10",
     "note": "Each score is verified at its source. 'Best' is our reading of the sources we opened; 2026 results are not yet published. All are on the 50 tasks of 2025, standard track unless noted.",
     "items": [
      {
       "value": 0.2599,
       "display": "Robot Learning Collective (independent), 2025-11: held-out Q 0.2599, success 0.1240; public Q 0.2605, success 0.1120. 1st place.",
       "level": "verified",
       "sources": [
        "s22",
        "s31",
        "s35"
       ],
       "note": "Built on π0.5. Uses task-specific correction rules (team report) and a set of four checkpoints (per the G0.5 report). Team budget about $13k.",
       "data": {
        "model": "Robot Learning Collective (π0.5-based)",
        "date": "2025-11",
        "avg": 25.99,
        "rl": false
       }
      },
      {
       "value": 0.2514,
       "display": "Comet (NVIDIA Research), 2025-11: held-out Q 0.2514, success 0.1140; public Q 0.1830. 2nd place.",
       "level": "verified",
       "sources": [
        "s22",
        "s32"
       ],
       "note": "The team says its public score came from an incomplete evaluation. Built on π0.5 with rejection-sampling fine-tuning on the policy's own successful simulator rollouts, which the authors distinguish from online RL.",
       "data": {
        "model": "Comet, NVIDIA (π0.5-based)",
        "date": "2025-11",
        "avg": 25.14,
        "rl": false
       }
      },
      {
       "value": 0.1591,
       "display": "SimpleAI Robot (Beijing Simple AI Technology), 2025-11: held-out Q 0.1591. 3rd place.",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "data": {
        "model": "SimpleAI Robot",
        "date": "2025-11",
        "avg": 15.91,
        "rl": false
       }
      },
      {
       "value": 0.1204,
       "display": "The North Star (Huawei CRI EAI Team), 2025-11: held-out Q 0.1204. 4th place.",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "data": {
        "model": "The North Star, Huawei",
        "date": "2025-11",
        "avg": 12.04,
        "rl": false
       }
      },
      {
       "value": 0.0947,
       "display": "Embodied Intelligence (independent), privileged-information track, 2025-11: held-out Q 0.0947.",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "note": "Not plotted: the privileged track lets the policy query the simulator."
      },
      {
       "value": 0.3453,
       "display": "Comet after the challenge, 2026-01: Q 0.3453 and success 0.15 on the validation instances, self-reported.",
       "level": "verified",
       "sources": [
        "s32",
        "s33"
       ],
       "note": "Added in v3 of the report (2026-01-05, comment 'Post-challenge bug fix'); v1 gave 0.22. The training loop returns the best checkpoint on validation, so the model was chosen on the instances it is scored on.",
       "data": {
        "model": "Comet post-challenge (π0.5-based)",
        "date": "2026-01",
        "avg": 34.53,
        "rl": false
       }
      },
      {
       "value": 0.2626,
       "display": "π0.5, one checkpoint trained 4 epochs, run by Galaxea, 2026-08: Q 0.2626 on 50 tasks x 10 instances.",
       "level": "verified",
       "sources": [
        "s35"
       ],
       "data": {
        "model": "π0.5 (Galaxea's run)",
        "date": "2026-08",
        "avg": 26.26,
        "rl": false
       }
      },
      {
       "value": 0.3136,
       "display": "G0.5 (Galaxea), one checkpoint, 2026-08: Q 0.3136 after 4 epochs and 0.2904 after 1 epoch; average of two evaluation runs; standard track, low-resolution RGB.",
       "level": "verified",
       "sources": [
        "s35"
       ],
       "note": "Compared in the same table with leaderboard public scores of 0.2605 (RLC) and 0.1830 (Comet). See issues.i3.",
       "data": {
        "model": "G0.5, Galaxea",
        "date": "2026-08",
        "avg": 31.36,
        "rl": false
       }
      },
      {
       "value": "2026 results pending",
       "display": "2026 scores were hidden on 2026-10-09 pending verification. Winners are due 2026-11-04.",
       "level": "verified",
       "sources": [
        "s23",
        "s16"
       ]
      }
     ],
     "short": "Q-score of 0.2599 on the 2025 held-out test",
     "chart": {
      "max": 100,
      "unit": "",
      "label": "Q score multiplied by 100"
     }
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, 'Copyright (c) 2023 Stanford Vision and Learning Group'.",
     "short": "MIT"
    },
    "license_data": {
     "value": [
      "MIT"
     ],
     "display": "Challenge demonstrations, task instances and the released 2025 hidden instances are labelled MIT on their Hugging Face cards. The 2026 raw HDF5 dataset has no card.",
     "level": "verified",
     "sources": [
      "s24",
      "s25",
      "s26",
      "s49"
     ],
     "checked": "2026-10-10",
     "note": "These files are rendered from, or point into, the encrypted asset bundle, which has its own licence (license_assets and issues.i6).",
     "short": "MIT (demonstrations and task instances)"
    },
    "license_assets": {
     "value": "BEHAVIOR Data Bundle EULA",
     "display": "Custom EULA, last revised 2022-12-08: non-commercial academic research only; the data is encrypted; use only inside OmniGibson; no reverse engineering; no redistribution of the key or of the data 'in whole or part'.",
     "level": "verified",
     "sources": [
      "s9",
      "s12",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Asset origin",
       "display": "Objects were mainly bought from TurboSquid and edited for simulation. Users get an encrypted copy and need not buy the assets.",
       "level": "verified",
       "sources": [
        "s12",
        "s2"
       ]
      },
      {
       "value": "Simulator licence (dependency)",
       "display": "The isaacsim 5.1.0.0 package is labelled 'NVIDIA Proprietary Software'. NVIDIA's docs say the Isaac Sim source on GitHub is Apache 2.0, while the Kit SDK and NVIDIA 3D assets fall under the Isaac Sim Additional Software and Materials License (use on systems with NVIDIA GPUs; no redistribution or modification). setup.sh links the general NVIDIA Software License Agreement instead.",
       "level": "verified",
       "sources": [
        "s40",
        "s41",
        "s42",
        "s9",
        "s48"
       ],
       "note": "The text we read of the Additional Software licence has no non-commercial clause."
      }
     ],
     "short": "A custom end-user licence (EULA) for non-commercial use. The assets are encrypted."
    },
    "access": {
     "value": "open",
     "display": "Code on GitHub. Demonstrations on ungated Hugging Face datasets. The encrypted asset bundle downloads after a click-through EULA in the installer. No account is needed.",
     "level": "verified",
     "sources": [
      "s7",
      "s9",
      "s11",
      "s24",
      "s25",
      "s28"
     ],
     "checked": "2026-10-10",
     "note": "setup.sh asks the user to accept the Conda terms, the NVIDIA Isaac Sim EULA and the BEHAVIOR Data Bundle EULA, or to pass flags that accept them.",
     "short": "Open, after accepting a click-through EULA"
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s9",
      "s8",
      "s24"
     ],
     "checked": "2026-10-10",
     "note": "Clause 2 of the asset EULA limits use to non-commercial academic research, and the tasks cannot run without the assets. Code (MIT) and demos (MIT) allow commercial use on their own. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "One task, run by the authors in 2022. The trained policy scored about 40% in simulation and 0% on a real Tiago++. A human-chosen sequence of actions scored about 22% on the real robot. No independent study.",
     "level": "inferred",
     "sources": [
      "s4",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Level is our mapping. Set-up: CollectTrash in a scanned digital twin of a lab mock apartment. The vision policy (RL-Prim.) ran 50 times in simulation (about 40%; Table 2 gives 0.42 for this policy) and 26 times on the real robot (0%). An 'optimal policy based on human input' ran 27 times on the real robot only (about 22%). So the often-quoted '22% real vs 40% sim' compares two different policies. In simulation, grasping was assisted (an object attaches when all fingers touch it); on the real robot grasping caused about 40% of failures, and 44% of the trained policy's errors came from wrong action choices due to image differences. Appendix G.2 of the 2024 version adds that three trained policies kept the same order in the real world as in simulation, with no statistic.",
     "short": "One task, tested by the authors only. About 40% in simulation and 0% on the real robot."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation."
    },
    "citations": {
     "value": 459,
     "display": "459 for the CoRL 2022 record (52 influential) and 191 for the 2024 arXiv record (20 influential), Semantic Scholar",
     "level": "verified",
     "sources": [
      "s30",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "Two separate records. Some papers cite both, so the counts should not be added.",
     "short": "459 and 191 for two separate records"
    },
    "github_stars": {
     "value": 1744,
     "display": "1,744 stars, 255 forks (StanfordVL/BEHAVIOR-1K)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "1,744"
    },
    "dataset_downloads": {
     "value": 80555,
     "display": "2025 challenge demos: 80,555 (Hub 'downloads' field), 1,776,633 all time, 38 likes. 2026 challenge demos: 70,031, 330,479 all time.",
     "level": "verified",
     "sources": [
      "s27",
      "s28"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API. We did not check the time window behind the 'downloads' field.",
     "short": "80,555 for the 2025 demonstrations on Hugging Face"
    },
    "used_by": {
     "value": "18 teams entered the 2025 challenge, from the US, China, Canada and South Korea. At least six papers from 2025 to 2026 report results on the challenge task set.",
     "level": "verified",
     "sources": [
      "s20",
      "s22",
      "s31",
      "s32",
      "s35",
      "s36",
      "s37",
      "s34"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Robot Learning Collective",
       "display": "Independent team of three, 2025-12 report. 1st place in 2025.",
       "level": "verified",
       "sources": [
        "s31",
        "s22"
       ]
      },
      {
       "value": "NVIDIA Research (Openpi Comet)",
       "display": "2025-12 report, 2nd place; post-challenge validation Q 0.3453.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "Galaxea (G0.5)",
       "display": "2026-08 technical report: Q 0.3136 on the 2025 task set with one checkpoint.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "Georgia Tech (MPVI)",
       "display": "2026-05: motion planning plus a VLA; reports a 113% higher mean Q than its run of the openpi-comet checkpoint.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "Shanghai Innovation Institute et al. (AHAT / TGPO)",
       "display": "2026-02, v2 2026-07: high-level planning success of 70.3% on the 50 challenge tasks, with plan feasibility checked by a PDDL solver and no robot control. Not comparable with Q-scores.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      },
      {
       "value": "Huawei Noah's Ark Lab, Canada",
       "display": "2026-04 analysis paper: re-ran the top two 2025 entries and proposed safety-aware scores.",
       "level": "verified",
       "sources": [
        "s34"
       ]
      },
      {
       "value": "2026 baselines",
       "display": "π0.5 (fork of OpenPI) and GR00T N1.7 (fork of Isaac-GR00T), with checkpoints for one task.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      }
     ],
     "short": "18 teams in 2025, and several reports in 2026"
    },
    "industry_use": {
     "value": [
      "NVIDIA",
      "Huawei",
      "Beijing Simple AI Technology",
      "Galaxea",
      "Simovation"
     ],
     "level": "verified",
     "sources": [
      "s22",
      "s32",
      "s19",
      "s34",
      "s35",
      "s14",
      "s20"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "NVIDIA",
       "display": "NVIDIA Research team placed 2nd in 2025. GR00T N1.7 is a 2026 baseline. NVIDIA sponsored the 2025 challenge.",
       "level": "verified",
       "sources": [
        "s22",
        "s32",
        "s19",
        "s20"
       ]
      },
      {
       "value": "Huawei",
       "display": "Huawei CRI EAI Team placed 4th in 2025. Noah's Ark Lab (Huawei Canada) published an analysis of the top entries.",
       "level": "verified",
       "sources": [
        "s22",
        "s34"
       ]
      },
      {
       "value": "Beijing Simple AI Technology",
       "display": "Placed 3rd in 2025.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "Galaxea",
       "display": "G0.5 technical report evaluates on the 2025 task set.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "Simovation",
       "display": "Provided the 2026 JoyLo teleoperation data in simulation; challenge sponsor.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "COHERENT",
      "Behavior-Skill",
      "ManiUnit"
     ],
     "level": "verified",
     "sources": [
      "s45",
      "s46",
      "s47"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list. None has a published sim-to-real study that we found.",
     "items": [
      {
       "value": "COHERENT",
       "display": "2024-09. Heterogeneous multi-robot benchmark built on the BEHAVIOR-1K platform; LLM planners.",
       "level": "verified",
       "sources": [
        "s45"
       ]
      },
      {
       "value": "Behavior-Skill",
       "display": "2026-08. Skill-level benchmark built on BEHAVIOR-1K demonstrations.",
       "level": "verified",
       "sources": [
        "s46"
       ]
      },
      {
       "value": "ManiUnit",
       "display": "2026-10. Manipulation skill dataset and benchmark built from 50 BEHAVIOR-1K activities.",
       "level": "verified",
       "sources": [
        "s47"
       ]
      }
     ],
     "short": "Skill and multi-robot benchmarks built on it"
    },
    "status": {
     "value": "active",
     "display": "Five releases from 2026-07 to 2026-10; the 2026 challenge closes 2026-10-16.",
     "level": "inferred",
     "sources": [
      "s6",
      "s7",
      "s14",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Repository pushed 2026-10-10. 311 open issues and pull requests (GitHub API).",
     "short": "Active. The 2026 challenge is running."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "The score ignores how the goal was reached",
     "text": "Q counts only the final states of the task's target objects. A 2026 analysis by Huawei Noah's Ark Lab (Canada) points out that dropping a non-target object, such as a chopping board, or hitting furniture is not penalised. It proposed safety-adjusted scores and re-ran the top two 2025 entries: adjusting for target-object violations cut task scores by up to 35%, and up to 40% of cases showed some violation when non-target objects were added. Averages fell from Q 0.256 to 0.239 for the winner and from 0.192 to 0.173 for the runner-up.",
     "level": "verified",
     "sources": [
      "s34",
      "s15"
     ],
     "status": "open",
     "short": "Only the final states of objects count, so unsafe actions are not penalised."
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "Re-running the same policy gives different per-task scores",
     "text": "The organisers state that the simulator is nondeterministic. The Huawei team re-ran the winner's published checkpoints with the official scripts: per-task scores differed from the posted ones by more than 0.27 on tasks 24, 26 and 27, while the winner's average stayed close to its posted 0.2605. In their re-runs the winner averaged Q 0.256 and the runner-up 0.192, against 0.2514 for the runner-up on the held-out test. A Georgia Tech paper ran an openpi-comet checkpoint on all 50 tasks and got per-task values averaging 0.0788 by our arithmetic, but it may not have used the competition's checkpoint suite. The winner also reports that its cloud evaluation machines rendered without NGX, which visibly degraded images but had very small impact on success rate.",
     "level": "verified",
     "sources": [
      "s15",
      "s34",
      "s22",
      "s36",
      "s31"
     ],
     "status": "open",
     "mitigation": {
      "text": "The 2026 rules forbid assembling results from repeated runs, require one rollout per instance, discourage tiled rendering, hide scores in the final week and have the organisers re-run top entries on hidden instances.",
      "sources": [
       "s15",
       "s16",
       "s17"
      ]
     },
     "note": "The 0.0788 mean is computed by us from MPVI's per-task table (Appendix A, values rounded to two decimals).",
     "short": "When another team re-ran the same checkpoints (saved versions of the trained policy), some task scores moved by more than 0.27."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "Papers compare numbers from different protocols",
     "text": "The winner's held-out 0.2599 used four checkpoints and task-specific correction rules. Comet's 0.3453 is a post-challenge score on the validation instances, with the checkpoint chosen on those instances. Galaxea's G0.5 report compares its own two-run average (10 instances per task; the set is not stated) with leaderboard public scores, including Comet's 0.1830, which the Comet team says came from an incomplete evaluation. AHAT reports 70.3% 'success' from planning alone. The 2025 hidden instances were published on 2025-12-15, so later papers can no longer report a held-out score on the 2025 set.",
     "level": "verified",
     "sources": [
      "s22",
      "s31",
      "s32",
      "s35",
      "s37",
      "s26"
     ],
     "status": "open",
     "short": "Later papers mix held-out scores from a hidden test set, public scores and scores that the authors selected themselves."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The software changed between challenge editions",
     "text": "The 2025 challenge ran on BEHAVIOR-1K v3.7 with Isaac Sim 4.5.0; the 2026 challenge runs on v3.9 with Isaac Sim 5.1.0. A maintainer says the first 50 tasks of the 2026 instances are the 2025 instances replayed on the newer version for compatibility. Users reported an unstable scene that gave invalid partial credit (issue #2324) and divergent action replay of the 2026 demos (issue #2344); maintainers could not reproduce either and measured 0.026 m of drift in one replay. The 2026 dataset was corrected on 2026-07-27 and 2026-08-24. 2025 and 2026 scores also differ in task count (50 against 100) and in robot rules.",
     "level": "verified",
     "sources": [
      "s10",
      "s9",
      "s38",
      "s39",
      "s16",
      "s15"
     ],
     "status": "open",
     "note": "NVIDIA's documentation marks Isaac Sim 5.1.0 as no longer supported (banner, checked 2026-10-10).",
     "short": "Scores from 2025 and 2026 come from different software versions and task sets."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "The tests use known tasks with new start states and allow hand-written rules",
     "text": "All 50 challenge tasks appear in both the training demonstrations and the test. The winning team calls the required generalisation very limited and notes there is no test of unseen object categories, language goals or new tasks. It also reports that training demonstrations were biased toward simpler instances. Its task-specific correction rules, such as reopening an accidentally closed gripper, raised Q 2.2 times on 13 tasks (39 episodes); a later paper declined to compare with it for that reason. The rules allow any method.",
     "level": "verified",
     "sources": [
      "s31",
      "s36",
      "s21",
      "s15"
     ],
     "status": "open",
     "short": "The tests reuse the training tasks. Hand-written rules raised the winning team's score."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Demonstrations labelled MIT are rendered from non-commercial assets",
     "text": "The challenge demonstrations are labelled MIT, but they are videos and states rendered from the encrypted asset bundle. That bundle's EULA allows only non-commercial academic research and forbids redistributing the data 'in whole or part'. The cards do not say how the two licences interact.",
     "level": "inferred",
     "sources": [
      "s24",
      "s25",
      "s9"
     ],
     "status": "open",
     "note": "Our reading of the licence texts. Not legal advice.",
     "short": "The MIT licence on the demonstrations and the non-commercial licence on the assets may conflict."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Evaluation is slow and needs NVIDIA RTX hardware",
     "text": "The Comet team reports that evaluating a single task can take from one hour to nearly a full day. On an RTX 4090, the organisers measured scene loading of about 150 to 300 seconds and 13.52 to 24.55 frames per second with random actions. The winner used an on-demand cluster and reports that a full evaluation finishes in under 2 days. Final 2026 evaluation runs on GPUs such as the RTX 3090, A5000 and TitanRTX.",
     "level": "verified",
     "sources": [
      "s32",
      "s15",
      "s31",
      "s17"
     ],
     "status": "open",
     "short": "Evaluating a single task on RTX GPUs can take hours."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A BEHAVIOR Challenge score shows how much of each known household task a simulated robot completes from new starting positions. Partial credit makes up a large part of the leading scores. It says little about a real home. The only real-robot test is one task from 2022, where the trained policy scored 0%.",
     "basis": [
      "facts.metric_detail",
      "facts.generalisation",
      "facts.sim_to_real",
      "issues.i5"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A score shows progress on known household tasks in simulation. It says little about real homes."
    },
    {
     "id": "r2",
     "text": "Do not read gaps of a few Q points (Q is the share of goal conditions met) between papers as real differences. Re-runs moved per-task scores by more than 0.27, and papers mix held-out, public and self-selected results from different software versions.",
     "basis": [
      "issues.i2",
      "issues.i3",
      "issues.i4",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Ignore small differences in Q-score between papers."
    },
    {
     "id": "r3",
     "text": "The benchmark is far from saturated, which means top scores are still far below the maximum. The best verified 2025 entry fully completed about one task run in eight (success 0.1240), so the benchmark can still separate strong systems.",
     "basis": [
      "facts.top_score"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Top scores are far below the maximum. The best verified entry fully completed about one task run in eight."
    },
    {
     "id": "r4",
     "text": "For companies, the asset licence matters more than the MIT code licence. The tasks need the encrypted, non-commercial asset bundle.",
     "basis": [
      "facts.license_assets",
      "facts.commercial_use",
      "issues.i6"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Commercial users should read the asset licence first."
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "CoRL 2022 paper (Section 6.2) and 2024 arXiv version (Section 6.2, Appendix G); BEHAVIOR site including Related Research (BRS, ACDC, MoMaGen, BEHAVIOR Vision Suite); 2025 and 2026 challenge pages; top-2 team reports (2512.06951, 2512.10071); G0.5 (2608.11739; real-robot results are on separate R1-Lite/R1-Pro tasks, not paired with BEHAVIOR scores); MPVI (2606.00985); Huawei analysis (2604.21192); Semantic Scholar citation contexts for both BEHAVIOR-1K records (650 citing papers) filtered for real-world mentions; web searches. No paired sim-vs-real study beyond the authors' 2022 one.",
     "date": "2026-10-10"
    },
    {
     "for": "challenge results report by the organisers",
     "where": "Challenge pages, Stanford HAI news (2025-09-22, a launch story), web search for a 2025 results or lessons paper. None found.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard (2026)",
     "where": "Hugging Face Space files data/results.jsonl (empty), data/self_reported_results.jsonl and the commit log; updates page. Verified 2026 results do not exist yet.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "setup.sh EULA text, Asset Sources page, 2024 paper Appendix D, Hugging Face cards, PyPI isaacsim metadata, Isaac Sim 5.1.0 licence pages, the NVIDIA Software License Agreement page linked by setup.sh.",
     "date": "2026-10-10"
    },
    {
     "for": "robots (R1 Pro maker)",
     "where": "Robots page, challenge pages and 2024 paper: maker not named. BRS paper and G0.5 report used for the inference.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "BEHAVIOR-1K: A Human-Centered, Embodied AI Benchmark with 1,000 Everyday Activities and Realistic Simulation (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2403.09227",
     "type": "paper",
     "publisher": "arXiv (Stanford University et al.)",
     "date": "2024-03-14",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "BEHAVIOR-1K, extended paper, full text v1 (Sections 4 to 7, Table 1, Appendices F and G)",
     "url": "https://arxiv.org/html/2403.09227v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-03-14",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "BEHAVIOR-1K: A Benchmark for Embodied AI with 1,000 Everyday Activities and Realistic Simulation (PMLR page)",
     "url": "https://proceedings.mlr.press/v205/li23a.html",
     "type": "paper",
     "publisher": "CoRL 2022, PMLR 205:80-93",
     "date": "2022-12",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "BEHAVIOR-1K, CoRL 2022 paper (PDF, Section 6.2)",
     "url": "https://proceedings.mlr.press/v205/li23a/li23a.pdf",
     "type": "paper",
     "publisher": "CoRL 2022, PMLR 205",
     "date": "2022-12",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "BEHAVIOR project site, home page",
     "url": "https://behavior.stanford.edu/",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "StanfordVL/BEHAVIOR-1K releases",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/releases",
     "type": "repo",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026-10-07",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: StanfordVL/BEHAVIOR-1K (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/StanfordVL/BEHAVIOR-1K",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "BEHAVIOR-1K LICENSE file",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "BEHAVIOR-1K setup.sh on main (licence prompts, Isaac Sim 5.1.0 install)",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/blob/main/setup.sh",
     "type": "repo",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026-08-27",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "BEHAVIOR-1K setup.sh at tag v3.7.2 (Isaac Sim 4.5.0)",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/blob/v3.7.2/setup.sh",
     "type": "repo",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2025-12-15",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "BEHAVIOR documentation: Installation (system requirements)",
     "url": "https://behavior.stanford.edu/getting_started/installation.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "BEHAVIOR documentation: Asset Sources",
     "url": "https://behavior.stanford.edu/behavior_components/asset_sources.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "BEHAVIOR documentation: Robots (OmniGibson)",
     "url": "https://behavior.stanford.edu/omnigibson/robots.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "2026 BEHAVIOR Challenge, home page",
     "url": "https://behavior.stanford.edu/challenge/index.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "2026 BEHAVIOR Challenge: Evaluation and Rules",
     "url": "https://behavior.stanford.edu/challenge/evaluation.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "2026 BEHAVIOR Challenge: Announcements / Updates",
     "url": "https://behavior.stanford.edu/challenge/updates.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026-10-09",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "2026 BEHAVIOR Challenge: Submission Guidelines",
     "url": "https://behavior.stanford.edu/challenge/submission.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "2026 BEHAVIOR Challenge: Dataset",
     "url": "https://behavior.stanford.edu/challenge/dataset.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "2026 BEHAVIOR Challenge: Baselines",
     "url": "https://behavior.stanford.edu/challenge/baselines.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "2025 BEHAVIOR Challenge (archive page)",
     "url": "https://behavior.stanford.edu/challenge/archive/2025/index.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "2025 BEHAVIOR Challenge: Evaluation and Rules (archive)",
     "url": "https://behavior.stanford.edu/challenge/archive/2025/evaluation.html",
     "type": "site",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Provisional 2025 BEHAVIOR Challenge leaderboard",
     "url": "https://behavior.stanford.edu/challenge/archive/2025/leaderboard.html",
     "type": "leaderboard",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2025-12-01",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "BEHAVIOR-1K 2026 Challenge leaderboard Space (README, data files, commit log)",
     "url": "https://huggingface.co/spaces/behavior-1k/2026-challenge-leaderboard",
     "type": "leaderboard",
     "publisher": "behavior-1k on Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "behavior-1k/2025-challenge-demos dataset card (meta/info.json)",
     "url": "https://huggingface.co/datasets/behavior-1k/2025-challenge-demos",
     "type": "dataset",
     "publisher": "behavior-1k on Hugging Face",
     "date": "2025-12-02",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "behavior-1k/2026-challenge-demos dataset card",
     "url": "https://huggingface.co/datasets/behavior-1k/2026-challenge-demos",
     "type": "dataset",
     "publisher": "behavior-1k on Hugging Face",
     "date": "2026-08-05",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "behavior-1k/2025-challenge-hidden-instances dataset (card and commit history)",
     "url": "https://huggingface.co/datasets/behavior-1k/2025-challenge-hidden-instances",
     "type": "dataset",
     "publisher": "behavior-1k on Hugging Face",
     "date": "2025-12-15",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Hugging Face Hub API record for behavior-1k/2025-challenge-demos (downloads)",
     "url": "https://huggingface.co/api/datasets/behavior-1k/2025-challenge-demos?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Hugging Face Hub API listing of behavior-1k datasets (2026 demos and raw data, created, modified, downloads)",
     "url": "https://huggingface.co/api/datasets?author=behavior-1k&limit=50",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Semantic Scholar API record for arXiv:2403.09227",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2403.09227?fields=title,citationCount,influentialCitationCount,externalIds,venue,year,publicationDate",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Semantic Scholar API search result for the CoRL 2022 paper (CorpusId 255198985)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/search?query=BEHAVIOR-1K%3A%20A%20Benchmark%20for%20Embodied%20AI%20with%201%2C000%20Everyday%20Activities%20and%20Realistic%20Simulation&fields=title,citationCount,influentialCitationCount,externalIds,venue,year&limit=5",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Task adaptation of Vision-Language-Action model: 1st Place Solution for the 2025 BEHAVIOR Challenge (v2)",
     "url": "https://arxiv.org/html/2512.06951v2",
     "type": "paper",
     "publisher": "arXiv (Robot Learning Collective)",
     "date": "2025-12-07",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Openpi Comet: Competition Solution For 2025 BEHAVIOR Challenge (v3, 'Post-challenge bug fix')",
     "url": "https://arxiv.org/html/2512.10071v3",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2026-01-05",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Openpi Comet, v1 of the report",
     "url": "https://arxiv.org/html/2512.10071v1",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-12-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "How VLAs (Really) Work In Open-World Environments",
     "url": "https://arxiv.org/html/2604.21192",
     "type": "paper",
     "publisher": "arXiv (Noah's Ark Laboratory, Huawei Technologies Canada)",
     "date": "2026-04-23",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "G0.5: One Autoregressive Stream for Robot Reasoning and Action (Section 5.3, Table 4)",
     "url": "https://arxiv.org/html/2608.11739",
     "type": "paper",
     "publisher": "arXiv (Galaxea)",
     "date": "2026-08-12",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Make Your VLA More Robust Without More Data By Interleaving Motion Planning (MPVI; Appendix A, Table 1)",
     "url": "https://arxiv.org/html/2606.00985",
     "type": "paper",
     "publisher": "arXiv (Georgia Institute of Technology)",
     "date": "2026-05-31",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Any House Any Task (v2 retitled TGPO: Trace-Guided Policy Optimization for Robot Task Planning), Table 1",
     "url": "https://arxiv.org/html/2602.12244",
     "type": "paper",
     "publisher": "arXiv (Shanghai Innovation Institute, Shanghai Jiao Tong University et al.)",
     "date": "2026-02-12",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "BEHAVIOR-1K issue #2324: invalid partial Q-score on putting_shoes_on_rack (Isaac Sim 5.1)",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/issues/2324",
     "type": "repo",
     "publisher": "StanfordVL/BEHAVIOR-1K (user report and maintainer replies)",
     "date": "2026-08-03",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "BEHAVIOR-1K issue #2344: action-only replay of 2026 demos on the 3.9 stack",
     "url": "https://github.com/StanfordVL/BEHAVIOR-1K/issues/2344",
     "type": "repo",
     "publisher": "StanfordVL/BEHAVIOR-1K (user report and maintainer replies)",
     "date": "2026-09-07",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "PyPI JSON metadata for isaacsim 5.1.0.0 (licence field)",
     "url": "https://pypi.org/pypi/isaacsim/5.1.0.0/json",
     "type": "index",
     "publisher": "PyPI (package by NVIDIA)",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "Isaac Sim 5.1.0 documentation: NVIDIA Isaac Sim Licensing",
     "url": "https://docs.isaacsim.omniverse.nvidia.com/5.1.0/common/licenses-isaac-sim.html",
     "type": "site",
     "publisher": "NVIDIA",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "Isaac Sim 5.1.0 documentation: NVIDIA Isaac Sim Additional Software and Materials License",
     "url": "https://docs.isaacsim.omniverse.nvidia.com/5.1.0/common/license-isaac-sim-additional.html",
     "type": "site",
     "publisher": "NVIDIA",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "BEHAVIOR Challenge Charts the Way Forward for Domestic Robotics",
     "url": "https://hai.stanford.edu/news/behavior-challenge-charts-the-way-forward-for-domestic-robotics",
     "type": "blog",
     "publisher": "Stanford HAI",
     "date": "2025-09-22",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "BEHAVIOR Robot Suite (BRS): Streamlining Real-World Whole-Body Manipulation for Everyday Household Activities",
     "url": "https://arxiv.org/html/2503.05652",
     "type": "paper",
     "publisher": "arXiv (Stanford University), CoRL 2025",
     "date": "2025-03-07",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "COHERENT: Collaboration of Heterogeneous Multi-Robot System with Large Language Models",
     "url": "https://arxiv.org/html/2409.15146",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-09-23",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "Behavior-Skill: A Fine-Grained Benchmark for Evaluating Vision-Language-Action Policies in Long-Horizon Tasks",
     "url": "https://arxiv.org/html/2608.30536",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-08-31",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "ManiUnit: A Manipulation Skill Dataset and Benchmark for Long-Horizon Tasks",
     "url": "https://arxiv.org/abs/2610.12089",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-10-08",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "NVIDIA Software License Agreement (page linked from setup.sh)",
     "url": "https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/",
     "type": "site",
     "publisher": "NVIDIA",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "behavior-1k/2026-challenge-rawdata README (returns 'Entry not found': no card)",
     "url": "https://huggingface.co/datasets/behavior-1k/2026-challenge-rawdata/raw/main/README.md",
     "type": "dataset",
     "publisher": "behavior-1k on Hugging Face",
     "date": "2026-06-22",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/core-sim-b.json. Folded the 2025 and 2026 BEHAVIOR Challenge into this record. Changes from the basic entry: embodiment narrowed to mobile-manipulator (bimanual-arm means a fixed base); sim_to_real level set to inferred with the full study design; '22% vs 40%' prior claim corrected (two different policies); added the 2025 hours conflict, Isaac Sim versions per edition, Semantic Scholar count for the CoRL record (459), validity v1, issues and readings. Checks ran on 2026-10-10 and into early 2026-10-11 local time."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy (the robot's control model) will do on a real robot.",
     "sub": "Only one task has been tested on a real robot, in 2022. The trained policy scored about 40% in simulation and 0% on the real robot.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether the robot does the task safely or carefully.",
     "sub": "The score counts only the final states of the task's target objects.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "How a policy handles new tasks or objects.",
     "sub": "The tests reuse the training tasks with new start states.",
     "basis": [
      "facts.generalisation",
      "issues.i5"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "BEHAVIOR-1K paper, real-robot study",
     "date": "2022-12",
     "by": "authors",
     "method": "The authors tested one task (CollectTrash) in a scanned copy of a lab apartment. The trained policy ran 50 times in simulation and 26 times on a real Tiago++ robot. Three trained policies were compared by rank.",
     "result": "The trained policy had about 40% success in simulation and 0% on the real robot. The 3 policies ranked in the same order. No statistic was reported.",
     "authors_view": "preliminary",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s4",
      "s2"
     ],
     "note": "A human-chosen 'optimal' policy scored about 22% on the real robot (27 runs) but was not scored in simulation. The rank-order result is in Appendix G.2 of the 2024 version only; its first sentence lists RL-Prim.Hist. twice."
    }
   ]
  },
  {
   "id": "bridgedata-v2",
   "name": "BridgeData V2",
   "full_name": "BridgeData V2: A Dataset for Robot Learning at Scale",
   "aliases": [
    "Bridge V2",
    "BridgeV2",
    "Bridge",
    "bridge_dataset",
    "bridge_orig",
    "Berkeley Bridge (OXE subset 'bridge')"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "A training dataset whose low-cost WidowX setup in toy kitchens is a common evaluation reference. Papers run real-robot tests in rebuilt Bridge scenes (OXE, Octo, OpenVLA, SpatialVLA, X-VLA), SimplerEnv simulates four Bridge tasks, AutoEval runs autonomous real Bridge cells, and the paper itself tests 6 methods including at a second lab.",
   "summary": {
    "text": "BridgeData V2 is a set of 60,096 robot trajectories recorded on a low-cost WidowX 250 arm, mostly in toy kitchens, with language labels. Its robot setup is widely reused for real-robot and simulated tests of generalist policies, but there is no fixed test protocol.",
    "short": "BridgeData V2 is a set of 60,096 robot trajectories (recorded robot runs) on a low-cost arm, mostly in toy kitchens. It is used as training data, and its robot setup is reused for tests.",
    "sources": [
     "s1",
     "s4",
     "s21"
    ]
   },
   "facts": {
    "kind": {
     "value": "dataset",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Presented as a dataset with training code and checkpoints. No fixed test set or scoring rule."
    },
    "kind_secondary": {
     "value": [
      "study"
     ],
     "display": "Also a one-off evaluation of 6 learning methods in the paper, including a test at a second lab",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "publishers": {
     "value": [
      "UC Berkeley",
      "Stanford University",
      "Carnegie Mellon University",
      "Google DeepMind"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "UC Berkeley",
       "display": "8 of 14 authors, including lead author Homer Walke and Sergey Levine (RAIL lab). Code copyright: Robotic AI & Learning Lab Berkeley.",
       "level": "verified",
       "sources": [
        "s2",
        "s6"
       ]
      },
      {
       "value": "Stanford University",
       "display": "4 authors (Kim, Du, Zhao, Finn)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Carnegie Mellon University",
       "display": "1 author (Zheng)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Google DeepMind",
       "display": "1 author (Vuong)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "13 of 14 authors are at universities; data collected at one institution (UC Berkeley)."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2023-08",
     "display": "arXiv v1 2023-08-24. Published at CoRL 2023.",
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Raw data files on the Berkeley server are dated 2023-06-20 (scripted) and 2023-08-20 (demonstrations). arXiv v3: 2024-01-17.",
     "short": "August 2023, at CoRL 2023"
    },
    "published_at": {
     "value": "CoRL 2023",
     "display": "Proceedings of the 7th Conference on Robot Learning, PMLR 229:1723-1736",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2024-03",
     "display": "Last official change: repo commit 2024-03-17 (RLDS conversion link). Data files unchanged since 2023-09. A Google-hosted copy (bridge_data_v2/0.0.1) appeared 2024-08-12 without documentation.",
     "level": "verified",
     "sources": [
      "s8",
      "s9",
      "s14"
     ],
     "checked": "2026-10-10",
     "short": "March 2024. A link in the repository was updated."
    },
    "version": {
     "value": "raw release + RLDS 1.0.0",
     "display": "Raw: demos_8_17.zip (411 GB) and scripted_6_18.zip (30 GB). Pre-processed RLDS copy bridge_dataset 1.0.0 at 256x256 (about 124 GB). No repo tags. Other copies carry other names and counts (items).",
     "level": "verified",
     "sources": [
      "s9",
      "s5",
      "s22"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Berkeley RLDS bridge_dataset 1.0.0",
       "display": "60,064 episodes: 53,192 train, 6,872 val (2023-09-21). OpenVLA renames it bridge_orig.",
       "level": "inferred",
       "sources": [
        "s10",
        "s22"
       ],
       "note": "Summed by us from shardLengths in dataset_info.json."
      },
      {
       "value": "OXE copy 'bridge' 0.1.0",
       "display": "28,935 episodes: 25,460 train, 3,475 test, 387.49 GiB. An early partial upload (OXE author, 2023-12).",
       "level": "verified",
       "sources": [
        "s12",
        "s13",
        "s16"
       ]
      },
      {
       "value": "Google bucket copy bridge_data_v2 0.0.1",
       "display": "60,063 episodes: 53,191 train, 6,872 val. Created 2024-08-12; not mentioned on the site, repo or OXE spreadsheet.",
       "level": "inferred",
       "sources": [
        "s14"
       ],
       "note": "Summed by us from dataset_info.json."
      },
      {
       "value": "Checkpoints",
       "display": "Released for GCBC, D-GCBC, LCBC, GCIQL and CRL (2023-09). The README says ACT and RT-1 checkpoints are not available.",
       "level": "verified",
       "sources": [
        "s11",
        "s5"
       ]
      }
     ],
     "short": "Raw zip files and an RLDS 1.0.0 copy. Several other copies exist."
    },
    "status": {
     "value": "dormant",
     "display": "No official change since March 2024. Use as training data and as a test setup is ongoing.",
     "level": "inferred",
     "sources": [
      "s7",
      "s8",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "31 open issues and PRs; no maintainer replies found in recent issue comments (latest user comment 2026-05-31).",
     "short": "No official changes since March 2024. Still widely used."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Every trajectory has a language label; the data also supports goal-image conditioning, which has no taxonomy value."
    },
    "generalisation": {
     "value": [
      "object-pose",
      "object-instance",
      "scene-layout",
      "visual"
     ],
     "display": "Paper tests: seen tasks with new object positions, distractors and lighting; unseen objects and environments; and a second lab with a different setup.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "New positions, objects, scenes and visuals"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper's evaluations ran on real WidowX arms. The setup is also simulated by SimplerEnv and replicated in world models (see validity)."
    },
    "simulator": {
     "value": "none",
     "display": "None. Real-robot data. Simulated copies of four Bridge tasks exist in SimplerEnv.",
     "level": "verified",
     "sources": [
      "s2",
      "s18"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "WidowX 250 6DOF",
     "display": "WidowX 250 6-DoF arm; fixed over-the-shoulder RGB-D camera, two RGB cameras moved every 50 trajectories, wrist camera; VR-controller teleoperation; 5 Hz control",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The paper puts the setup cost at about $4,000. Most data has only the fixed camera view.",
     "short": "WidowX 250"
    },
    "scene": {
     "value": [
      "kitchen",
      "tabletop"
     ],
     "display": "Toy kitchens on tables (7 of 24 environments, most of the data), tabletops, standalone toy sinks, a toy laundry machine",
     "level": "verified",
     "sources": [
      "s4",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "'Kitchen' here means toy kitchens, not real kitchens."
    },
    "tasks": {
     "value": 13,
     "display": "13 skills (e.g. pick-and-place, pushing, wiping, folding, stacking, sweeping, opening doors and drawers). The paper gives no separate task count.",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "A skill is 'a group of trajectories that require similar motions' (Table 5). Not comparable with task counts of other datasets.",
     "short": "13 skills"
    },
    "scenes": {
     "value": 24,
     "display": "24 environments in 4 categories",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "24 environments"
    },
    "objects": {
     "value": 100,
     "display": "More than 100 objects",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Section 3.3 gives 'more than 100'; no exact count."
    },
    "demonstrations": {
     "value": 60096,
     "display": "60,096 trajectories: 50,365 teleoperated demonstrations and 9,731 scripted pick-and-place rollouts (84% human, 16% scripted). The PMLR abstract says 53,896.",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s4",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT: arXiv v1-v3 and the site say 60,096; the CoRL/PMLR abstract says 53,896. Released RLDS copies hold 60,064 (Berkeley) and 60,063 (Google) episodes.",
     "items": [
      {
       "value": "Average length 38 steps at 5 Hz",
       "display": "640x480 images; crowdsourced language labels added after collection",
       "level": "verified",
       "sources": [
        "s4",
        "s2"
       ]
      },
      {
       "value": "All-zero first action",
       "display": "OpenVLA reports every demonstration starts with an all-zero action; training without removing it made policies freeze.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      }
     ],
     "short": "60,096 trajectories"
    },
    "scale": {
     "value": "441 GB raw",
     "display": "Raw: 411 GB demonstrations + 30 GB scripted (JPEG, PNG, pkl). RLDS at 256x256: about 124 GB.",
     "level": "verified",
     "sources": [
      "s9",
      "s22"
     ],
     "checked": "2026-10-10",
     "short": "441 GB of raw data"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Later papers sometimes give partial credit (0.5) for some tasks, e.g. OpenVLA."
    },
    "metric_detail": {
     "value": "Success rate on paper-defined tasks",
     "display": "Paper: 8 seen and 6 unseen tasks, 10 trials each, 6 methods (GCBC, D-GCBC, ACT, CRL, LCBC, RT-1). Later papers define their own Bridge-setup tasks.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper results",
       "display": "Seen tasks average: GCBC 0.49, D-GCBC 0.49, ACT 0.41, CRL 0.42, LCBC 0.23, RT-1 0.49. Unseen: 0.60, 0.55, 0.28, 0.52, 0.08, 0.50.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Later real Bridge-setup suites",
       "display": "OXE: Bridge tasks at two labs. OpenVLA: 17 tasks x 10 trials. SpatialVLA: 7 task suites. X-VLA: 5 tasks. AutoEval: 5 tasks x 50 rollouts. Each defines its own tasks and objects.",
       "level": "verified",
       "sources": [
        "s25",
        "s21",
        "s26",
        "s27",
        "s19"
       ]
      }
     ],
     "short": "Success rate. Each paper sets its own tasks."
    },
    "trials": {
     "value": "10 per task (paper); varies later",
     "display": "Paper: 10 trials per task. OpenVLA: 10 per task (170 per policy). SimplerEnv real: 24 per task. AutoEval: 50 per policy and task.",
     "level": "verified",
     "sources": [
      "s2",
      "s21",
      "s18",
      "s19"
     ],
     "checked": "2026-10-10",
     "short": "10 per task in the paper"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "The paper reports averages over 10 trials without error bars. OpenVLA reports standard errors for its Bridge results."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s21",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "Each paper runs its own trials. AutoEval, a separate project, offers public autonomous Bridge cells, but its results go back to the submitter."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the site or repo."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6",
      "s7",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, Copyright (c) 2023 Robotic AI & Learning Lab Berkeley. The robot controller repo (bridge_data_robot) is also MIT."
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "display": "The site states all data is provided under CC BY 4.0.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "No licence file on the data server. Third-party copies on Hugging Face carry other labels (items); they do not change the original terms.",
     "items": [
      {
       "value": "apache-2.0",
       "display": "IPEC-COMMUNITY/bridge_orig_lerobot (91,062 downloads, Hub 'downloads' field)",
       "level": "reported",
       "sources": [
        "s37"
       ]
      }
     ],
     "short": "CC BY 4.0"
    },
    "license_assets": {
     "value": "not applicable",
     "level": "inferred",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Real-robot recordings only; no 3D assets are distributed."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s9",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Direct download from the Berkeley server, no registration.",
     "short": "Open download"
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s4",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "MIT code and CC BY 4.0 data both allow commercial use with attribution. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "display": "The paper's scores come from real robots. Simulated and learned copies of the Bridge setup have been checked against real robots (see validity).",
     "level": "inferred",
     "sources": [
      "s2",
      "s18",
      "s19",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "SimplerEnv's WidowX scenes matched real results for 3 policies (Pearson r 0.575 to 1.000 per task). AutoEval re-tested SIMPLER's Bridge scenes with 6 policies and found it policy-dependent (mean r 0.548 by our calculation). WorldGym, a world model, matched OpenVLA's real Bridge trials at r 0.78. Offline action error on Bridge validation data correlated negatively with real success in two studies. RobotArena ∞ found every policy scored higher on SIMPLER's 4 scenes than on 70 Bridge-derived scenes, but measured no real correlation.",
     "short": "Scores come from real robots. Studies of simulated copies disagree."
    },
    "real_reproducibility": {
     "value": "multi-site-measured",
     "display": "Small: two two-lab comparisons with the same policies (Bridge paper, 6 methods; OXE, 4 models).",
     "level": "inferred",
     "sources": [
      "s2",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "Bridge paper: 3 tasks at the authors' lab and an unnamed second lab, zero-shot; averages Lab 1 to Lab 2: GCBC 0.30 to 0.13, D-GCBC 0.23 to 0.13, ACT 0.03 to 0.10, CRL 0.13 to 0.20, LCBC 0.13 to 0.03, RT-1 0.47 to 0.40 (Pearson r 0.775 by our calculation). OXE: Stanford IRIS vs Berkeley RAIL, r 0.89 over 4 models (our calculation). Other papers rebuild the setup approximately; OpenVLA could not buy the original objects."
    },
    "citations": {
     "value": 912,
     "display": "912 (Semantic Scholar; 91 influential)",
     "level": "verified",
     "sources": [
      "s31"
     ],
     "checked": "2026-10-10",
     "short": "912"
    },
    "github_stars": {
     "value": 292,
     "display": "292 stars, 35 forks (rail-berkeley/bridge_data_v2)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "292"
    },
    "used_by": {
     "value": "At least 8 model reports test in the Bridge setup or train on the data",
     "level": "verified",
     "sources": [
      "s25",
      "s23",
      "s21",
      "s26",
      "s27",
      "s28",
      "s32",
      "s33"
     ],
     "checked": "2026-10-10",
     "note": "A lower bound from reports we opened.",
     "items": [
      {
       "value": "RT-1-X, RT-2-X",
       "display": "2023-10. Real Bridge tests at Stanford IRIS and Berkeley RAIL; Bridge data gave RT-2-X new skills on the Google Robot.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Octo",
       "display": "2024-05. Zero-shot WidowX BridgeV2 tests.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "OpenVLA",
       "display": "2024-06. 17 Bridge tasks x 10 trials: OpenVLA 70.6 ± 3.2%, RT-2-X 50.6 ± 3.5%, Octo 20.0 ± 2.6%, RT-1-X 18.5 ± 2.7%.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "SpatialVLA",
       "display": "2025-01. Real WidowX tests in the BridgeV2 setup, 7 task suites.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "OpenVLA-OFT",
       "display": "2025-02. Scales its recipe to BridgeData V2 on a real arm (appendix).",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "X-VLA",
       "display": "2025-10. Real tests following 'the BridgeData-v2 benchmark', 5 tasks.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "π0, GR00T N1",
       "display": "Physical Intelligence (2024-10) and NVIDIA (2025-03) include Bridge V2 in pretraining.",
       "level": "verified",
       "sources": [
        "s32",
        "s33"
       ]
      }
     ],
     "short": "At least 8 model reports"
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Physical Intelligence",
      "NVIDIA",
      "Microsoft"
     ],
     "level": "verified",
     "sources": [
      "s25",
      "s32",
      "s33",
      "s35",
      "s36"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "RT-X real tests in the Bridge setup.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Physical Intelligence",
       "display": "Bridge v2 in the π0 pretraining mixture.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Bridge-v2 in GR00T N1 pretraining; a GR00T N1.7 checkpoint for SimplerEnv Bridge; Cosmos-Reason1 uses BridgeData V2 for training and 100 benchmark questions.",
       "level": "verified",
       "sources": [
        "s33",
        "s34",
        "s35"
       ]
      },
      {
       "value": "Microsoft",
       "display": "Microsoft Research collected 822 WidowX episodes in a BridgeData V2-compatible setup (bridge_data_msr).",
       "level": "verified",
       "sources": [
        "s36"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "SimplerEnv (WidowX)",
      "AutoEval",
      "WorldGym",
      "RobotArena ∞ BridgeSim"
     ],
     "display": "Evaluations built on the Bridge setup or data",
     "level": "verified",
     "sources": [
      "s18",
      "s19",
      "s20",
      "s29"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "SimplerEnv (WidowX)",
       "display": "2024-05. Four simulated Bridge tasks. Has its own record.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "AutoEval",
       "display": "2025-03. Autonomous real WidowX cells (drawer, sink, cloth). Has its own record.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "WorldGym",
       "display": "2025-05. World model run on OpenVLA's Bridge task suite.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "RobotArena ∞ BridgeSim",
       "display": "2025-10. 70 simulated environments rebuilt from Bridge V2 videos.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      }
     ],
     "short": "4 evaluations built on it"
    }
   },
   "validity": [
    {
     "id": "v1",
     "name": "SimplerEnv (simulated WidowX + Bridge)",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 3 policies (RT-1-X, Octo-Base and Octo-Small) ran 4 Bridge tasks in simulation and on a real WidowX arm, with 24 real trials per task.",
     "result": "Pearson correlation r = 0.827, 0.575, 1.000 and 0.990 for success on the 4 tasks. MMRV (a measure of how often two rankings disagree) was 0 on 3 tasks and 0.111 on one.",
     "authors_view": "strong correlation",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s18"
     ],
     "note": "Table V. 4 of 16 SimplerEnv authors are BridgeData V2 authors (Walke, Vuong, Finn, Levine). Mean r over the 4 tasks is 0.848 by our arithmetic."
    },
    {
     "id": "v2",
     "name": "Offline action error on Bridge data (SimplerEnv baseline)",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 3 policies were ranked by action error on 25 Bridge validation trajectories, and the ranking was compared with real success on 4 tasks.",
     "result": "Pearson r = -0.951, -0.342, -0.857 and -1.000 on the 4 tasks. MMRV was 0.389, 0.194, 0.125 and 0.366.",
     "authors_view": "not a good proxy",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s18"
     ],
     "note": "Table XII."
    },
    {
     "id": "v3",
     "name": "AutoEval re-test of SIMPLER Bridge scenes",
     "date": "2025-03",
     "by": "authors",
     "method": "The same 6 policies ran 4 Bridge tasks in SIMPLER and in real trials run by humans, with 50 rollouts (test runs) each.",
     "result": "Pearson r = 0.548 as the mean of 4 tasks, with a range of 0.131 to 0.942. MMRV was 0.207.",
     "authors_view": "policy dependent",
     "n_policies": 6,
     "level": "inferred",
     "sources": [
      "s19"
     ],
     "note": "Computed by us from Tables 2 and 3; the values match the bars in Figure 7. Example: Open-π0 put eggplant in sink, 6/50 in SIMPLER vs 47/50 real. Overlap with Bridge authors: Levine."
    },
    {
     "id": "v4",
     "name": "Offline action error on Bridge data (AutoEval baseline)",
     "date": "2025-03",
     "by": "authors",
     "method": "Policies were ranked by action error on 400 Bridge validation trajectories, and the ranking was compared with real success in human-run trials on 5 tasks.",
     "result": "Pearson r of about -0.26, as the mean of 5 tasks",
     "authors_view": "negatively correlates",
     "level": "inferred",
     "sources": [
      "s19"
     ],
     "note": "Read by us from Figure 7 bar heights; the text states only the sign. We could not reproduce the bars from Table 4, which lists a different policy set (it includes GCBC, which has no real result)."
    },
    {
     "id": "v5",
     "name": "WorldGym (world model)",
     "date": "2025-05",
     "by": "independent",
     "method": "RT-1-X, Octo and OpenVLA were run in a world model (a model that predicts what the camera will see next) from the first frames of OpenVLA's 170 real Bridge trials on 17 tasks. The results were compared task by task.",
     "result": "Pearson r = 0.78 for per-task success. The policies' mean scores were within 3.3 points of the real ones on average.",
     "authors_view": "highly correlate",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s20"
     ]
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How one paper's numbers compare with another paper's numbers.",
     "sub": "Each paper rebuilds its own toy-kitchen tasks.",
     "basis": [
      "facts.metric_detail",
      "issues.i4"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a low error on recorded data means success on the real robot.",
     "sub": "The validation error of a policy (the robot's control model) correlated negatively with its real success. Validation error measures how far the policy's actions are from the recorded ones.",
     "basis": [
      "issues.i6"
     ]
    },
    {
     "id": "l3",
     "text": "How a policy handles work outside toy kitchens.",
     "sub": "The data comes from one lab and one low-cost arm, and the tasks need little precision.",
     "basis": [
      "facts.scene",
      "issues.i8"
     ]
    }
   ],
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Several copies exist with different sizes and names",
     "text": "The paper and site give 60,096 trajectories; the PMLR abstract gives 53,896. Released copies: Berkeley RLDS bridge_dataset 1.0.0 with 60,064 episodes (renamed bridge_orig by OpenVLA), the OXE copy 'bridge' with 28,935 episodes, and an undocumented Google copy bridge_data_v2 0.0.1 with 60,063. Papers that say they trained on 'Bridge' may mean any of these.",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s10",
      "s12",
      "s14",
      "s22",
      "s16"
     ],
     "status": "open",
     "short": "The released copies hold from 28,935 to 60,064 episodes and have different names."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "Every demonstration starts with an all-zero action",
     "text": "OpenVLA found that each demonstration records an all-zero action at the first step. Training without removing it gave policies that froze. RT-2-X was trained without this filtering; OpenVLA says the RT-2-X developers queried the second-most-likely action in the OXE Bridge evaluations, which the OXE paper does not mention.",
     "level": "verified",
     "sources": [
      "s21",
      "s25"
     ],
     "status": "open",
     "mitigation": {
      "text": "OpenVLA drops the first transition of every demonstration; this was enough to stop most freezing.",
      "sources": [
       "s21"
      ]
     },
     "short": "Policies trained without removing the all-zero first action froze."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Crowdsourced language labels contain errors",
     "text": "Labels were added after collection through a crowdsourcing platform. A user listed mismatched instructions on the OXE tracker; an OXE author replied the labels were crowdsourced and 'there is some label noise', and the updated Berkeley copy has the same noise.",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "status": "open",
     "short": "Some language labels do not match what the robot did."
    },
    {
     "id": "i4",
     "type": "inconsistent-reporting",
     "title": "There is no standard real-robot test",
     "text": "Each paper builds its own Bridge-setup tasks: the Bridge paper 14 tasks, OXE its own, OpenVLA 17, SpatialVLA 7 suites, X-VLA 5, SimplerEnv 4, AutoEval 5. OpenVLA reproduced the sink scene with 'rough approximations' and could not buy the original objects. RT-1-X scored 27% in OXE's Bridge test, 18.5% in OpenVLA's, and 0 to 4% success on SimplerEnv's real tasks.",
     "level": "verified",
     "sources": [
      "s2",
      "s25",
      "s21",
      "s26",
      "s27",
      "s18",
      "s19"
     ],
     "status": "open",
     "short": "Papers use different tasks and rebuilt scenes, so their scores cannot be compared directly."
    },
    {
     "id": "i5",
     "type": "shortcut",
     "title": "A policy trained beside the simulated Bridge test can match top scores",
     "text": "SimplerEnv's WidowX test is meant to be passed by policies trained on real BridgeData V2, but it does not restrict training data. A 2026 audit trained a 22M-parameter policy per task on 120 scripted demonstrations recorded in simulation beside the test and reached 94.8% (91/96), against 95.8% (92/96) for X-VLA. The audit says a high score alone is not evidence of capability on this test.",
     "level": "verified",
     "sources": [
      "s30"
     ],
     "status": "open",
     "short": "A small policy with 22 million parameters, trained on data recorded beside the simulated test, nearly matches the best score."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Offline action error is a poor predictor of real success",
     "text": "Ranking policies by action error on Bridge validation data gave negative correlations with real success in SimplerEnv (r -0.342 to -1.000 on 4 tasks) and AutoEval (mean about -0.26).",
     "level": "verified",
     "sources": [
      "s18",
      "s19"
     ],
     "status": "open",
     "short": "Lower validation error did not mean higher real success."
    },
    {
     "id": "i7",
     "type": "protocol-variance",
     "title": "Simulated proxies disagree with each other",
     "text": "SimplerEnv's own study found close agreement with real WidowX results for 3 policies. AutoEval, with 6 policies, found SIMPLER's accuracy depends on the policy (mean r 0.548 by our calculation). RobotArena ∞ found all tested policies score much higher on SIMPLER's 4 scenes than on its 70 Bridge-derived scenes and says SIMPLER may overestimate performance; it has no real-robot comparison.",
     "level": "verified",
     "sources": [
      "s18",
      "s19",
      "s29"
     ],
     "status": "contested",
     "counter": {
      "text": "SimplerEnv reports MMRV 0 on 3 of 4 Bridge tasks and Pearson r up to 1.000 for its 3 policies.",
      "sources": [
       "s18"
      ],
      "short": "SimplerEnv's own study found near-perfect ranking agreement with real results on 3 of 4 tasks."
     },
     "short": "Studies disagree on how well simulated Bridge scenes, used as stand-ins for real tests, track real results."
    },
    {
     "id": "i8",
     "type": "other",
     "title": "The authors say the data collection is narrow",
     "text": "The authors list as limits that tasks are 'generally low-precision', that data comes from a single institution, and that others may find it hard to standardise on the same robot.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "The data comes from a single lab and one robot type, and the tasks are low-precision."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "'Tested on Bridge' names a robot setup. It does not name a fixed test. Two papers' Bridge numbers are only comparable if they used the same tasks, objects and trial counts, which is rare.",
     "basis": [
      "facts.metric_detail",
      "issues.i4"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Bridge results from different papers are not comparable."
    },
    {
     "id": "r2",
     "text": "The low-cost, simple setup is the reason it became a shared reference. It is also its limit, because toy kitchens and low-precision tasks say little about harder real-world work.",
     "basis": [
      "facts.robots",
      "issues.i8",
      "facts.used_by"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The setup is easy to reproduce, but the tasks are simple."
    },
    {
     "id": "r3",
     "text": "Simulated Bridge scores (SimplerEnv) are weak evidence about real performance. The paired studies are small and they disagree. The simulated test can also be matched by training close to it.",
     "basis": [
      "issues.i5",
      "issues.i7",
      "facts.sim_to_real"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Treat simulated Bridge scores as weak evidence about real performance."
    }
   ],
   "searched": [
    {
     "for": "validity / sim_to_real",
     "where": "Bridge paper v3 (no sim experiments); SimplerEnv Tables V, XII; AutoEval v2 Tables 1-4 and Figure 7; WorldGym v3; RobotArena ∞ (no real correlation); OXE paper; 2026 audit 2606.04233 (Section 6); IRASim, WorldEval and dWorldEval leads (WorldEval and dWorldEval do not use Bridge).",
     "date": "2026-10-10"
    },
    {
     "for": "tasks, objects",
     "where": "Paper Sections 3.2-3.3 and Table 5; site. No task count beyond 13 skills; objects given as 'more than 100'.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "No single headline score exists for the dataset; real Bridge suites differ by paper. SimplerEnv WidowX scores belong to the SimplerEnv record.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "BridgeData V2: A Dataset for Robot Learning at Scale (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2308.12952",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-08",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "BridgeData V2 paper, full text v3",
     "url": "https://arxiv.org/html/2308.12952v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-01",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "BridgeData V2, PMLR 229 (CoRL 2023) page",
     "url": "https://proceedings.mlr.press/v229/walke23a.html",
     "type": "paper",
     "publisher": "PMLR",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "BridgeData V2 project site",
     "url": "https://rail-berkeley.github.io/bridgedata/",
     "type": "site",
     "publisher": "UC Berkeley RAIL",
     "date": "2023-08",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "bridge_data_v2 README",
     "url": "https://github.com/rail-berkeley/bridge_data_v2",
     "type": "repo",
     "publisher": "UC Berkeley RAIL",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "bridge_data_v2 LICENSE",
     "url": "https://github.com/rail-berkeley/bridge_data_v2/blob/main/LICENSE",
     "type": "repo",
     "publisher": "UC Berkeley RAIL",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: rail-berkeley/bridge_data_v2",
     "url": "https://api.github.com/repos/rail-berkeley/bridge_data_v2",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "bridge_data_v2 commit history",
     "url": "https://github.com/rail-berkeley/bridge_data_v2/commits/main",
     "type": "repo",
     "publisher": "UC Berkeley RAIL",
     "date": "2024-03-17",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Berkeley data server listing (raw zips, tfds folder)",
     "url": "https://rail.eecs.berkeley.edu/datasets/bridge_release/data/",
     "type": "dataset",
     "publisher": "UC Berkeley RAIL",
     "date": "2023-09",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "Berkeley RLDS bridge_dataset 1.0.0 dataset_info.json",
     "url": "https://rail.eecs.berkeley.edu/datasets/bridge_release/data/tfds/bridge_dataset/1.0.0/dataset_info.json",
     "type": "dataset",
     "publisher": "UC Berkeley RAIL",
     "date": "2023-09-21",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Released checkpoints listing",
     "url": "https://rail.eecs.berkeley.edu/datasets/bridge_release/checkpoints/",
     "type": "dataset",
     "publisher": "UC Berkeley RAIL",
     "date": "2023-09",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "TensorFlow Datasets catalog: bridge (OXE copy)",
     "url": "https://www.tensorflow.org/datasets/catalog/bridge",
     "type": "dataset",
     "publisher": "Google",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "OXE bridge 0.1.0 dataset_info.json",
     "url": "https://storage.googleapis.com/gresearch/robotics/bridge/0.1.0/dataset_info.json",
     "type": "dataset",
     "publisher": "Google",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "bridge_data_v2 0.0.1 dataset_info.json in gs://gresearch/robotics",
     "url": "https://storage.googleapis.com/gresearch/robotics/bridge_data_v2/0.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "Google",
     "date": "2024-08-12",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "OXE issue #30: OXE Bridge copy is an early upload (author reply)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/30",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "OXE issue #15: Bridge label noise (author reply)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/15",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2023-11",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "SimplerEnv: Evaluating Real-World Robot Manipulation Policies in Simulation (Tables V, XII; Appendix B)",
     "url": "https://arxiv.org/html/2405.05941",
     "type": "paper",
     "publisher": "arXiv (CoRL 2024)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World (v2; Tables 1-4, Figure 7)",
     "url": "https://arxiv.org/html/2503.24278v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "WorldGym: World Model as An Environment for Policy Evaluation (v3)",
     "url": "https://arxiv.org/html/2506.00613v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "OpenVLA (Section 5.1; Appendices B.1, C)",
     "url": "https://arxiv.org/html/2406.09246",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "OpenVLA README (download Berkeley RLDS copy, 124 GB, rename to bridge_orig)",
     "url": "https://github.com/openvla/openvla/blob/main/README.md",
     "type": "repo",
     "publisher": "OpenVLA authors",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Octo: An Open-Source Generalist Robot Policy",
     "url": "https://arxiv.org/html/2405.12213",
     "type": "paper",
     "publisher": "arXiv (RSS 2024)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Open X-Embodiment paper v9 (Table I Bridge results; Table II)",
     "url": "https://arxiv.org/html/2310.08864v9",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "SpatialVLA (real WidowX BridgeV2 evaluation)",
     "url": "https://arxiv.org/html/2501.15830",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "X-VLA (real-world BridgeData-v2 benchmark tasks)",
     "url": "https://arxiv.org/html/2510.10274",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "OpenVLA-OFT: Fine-Tuning Vision-Language-Action Models (Appendix, BridgeData V2)",
     "url": "https://arxiv.org/html/2502.19645",
     "type": "paper",
     "publisher": "arXiv (RSS 2025)",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "RobotArena ∞: Scalable Robot Benchmarking via Real-to-Sim Translation",
     "url": "https://arxiv.org/html/2510.23571",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (Section 6, data-source dependence)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Semantic Scholar API record for arXiv:2308.12952",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2308.12952?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "π0 paper (Bridge v2 in pretraining mixture)",
     "url": "https://arxiv.org/html/2410.24164",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "GR00T N1 paper (Bridge-v2 among OXE subsets)",
     "url": "https://arxiv.org/html/2503.14734",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Isaac-GR00T README (GR00T-N1.7-SimplerEnv-Bridge checkpoint)",
     "url": "https://github.com/NVIDIA/Isaac-GR00T",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Cosmos-Reason1 (BridgeData V2 in training data and benchmark)",
     "url": "https://arxiv.org/html/2503.15558v3",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "bridge_data_msr dataset_info.json (Microsoft Research)",
     "url": "https://storage.googleapis.com/gresearch/robotics/bridge_data_msr/0.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "Google (bucket); Microsoft Research (data)",
     "date": "2024-04-11",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Hugging Face: IPEC-COMMUNITY/bridge_orig_lerobot (third-party copy)",
     "url": "https://huggingface.co/datasets/IPEC-COMMUNITY/bridge_orig_lerobot",
     "type": "secondary",
     "publisher": "Third party",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "GitHub API: rail-berkeley/bridge_data_robot (MIT)",
     "url": "https://api.github.com/repos/rail-berkeley/bridge_data_robot",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the checked basic entry and research/raw/inventory/core-real.json. Added: episode counts of every copy, all-zero first action, label noise, checkpoint gaps, five validity studies (two recomputed or read from figures), two-lab measurements with our correlation figures, 2026 audit finding on SimplerEnv WidowX, adoption. Corrected the basic entry: at Lab 2, two of six methods improved (ACT, CRL)."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "calvin",
   "name": "CALVIN",
   "full_name": "CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks",
   "aliases": [
    "CALVIN (Composing Actions from Language and Vision)",
    "CALVIN ABC→D",
    "CALVIN ABCD→D",
    "CALVIN D→D",
    "LH-MTLC",
    "MTLC"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "CALVIN is a simulated benchmark in which one Franka Panda arm must complete chains of five language instructions at a desk in four similar environments. It is scored by how many instructions in a row a policy completes, out of 5, over 1,000 fixed chains.",
    "short": "CALVIN is a simulated benchmark in which a robot arm at a desk follows chains of five spoken-style instructions. It tests how many of the five tasks a policy (the robot's control model) completes in a row.",
    "sources": [
     "s2",
     "s4"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents CALVIN as a benchmark with an environment, a dataset and a challenge with fixed evaluation protocols."
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "Also a dataset of about 24 hours of teleoperated play data",
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The paper lists three components: the CALVIN environment, the CALVIN dataset and the CALVIN Challenge."
    },
    "publishers": {
     "value": [
      "University of Freiburg",
      "University of Technology Nuremberg"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Oier Mees and Lukas Hermann (equal contribution), Erick Rosete-Beas, Wolfram Burgard. Funded by the German Federal Ministry of Education and Research (contract 01IS18040B-OML).",
     "items": [
      {
       "value": "University of Freiburg",
       "display": "Autonomous Intelligent Systems Lab. Affiliation of Oier Mees, Lukas Hermann and Erick Rosete-Beas.",
       "level": "verified",
       "sources": [
        "s2",
        "s4"
       ]
      },
      {
       "value": "University of Technology Nuremberg",
       "display": "Affiliation of Wolfram Burgard on the paper.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Both affiliations are universities."
    },
    "region": {
     "value": "europe",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Both institutions are in Germany."
    },
    "first_release": {
     "value": "2021-12",
     "display": "arXiv v1 on 2021-12-06. Published in IEEE Robotics and Automation Letters, vol. 7, no. 3, pp. 7327-7334 (July 2022).",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s3",
      "s5"
     ],
     "checked": "2026-10-11",
     "note": "arXiv v4 (2022-07-13) is the accepted version (received 2022-02-23, accepted 2022-05-22). The README states that CALVIN won the 2022 RA-L Best Paper Award. The GitHub repository was created on 2021-07-20 (GitHub API).",
     "short": "December 2021. Published in RA-L in 2022."
    },
    "published_at": {
     "value": "IEEE Robotics and Automation Letters 7(3), 2022",
     "level": "verified",
     "sources": [
      "s3",
      "s2"
     ],
     "checked": "2026-10-11",
     "note": "DOI 10.1109/LRA.2022.3180108.",
     "short": "RA-L 2022"
    },
    "latest_update": {
     "value": "2025-09",
     "display": "2025-09-08: FLOWER added to the README model list and the website leaderboard (website Last-Modified header 2025-09-08). Last change to the evaluation code: 2023-12-07. Last dataset file change: the D→D zip, 2023-02-23.",
     "level": "verified",
     "sources": [
      "s9",
      "s4",
      "s10",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Commit history: 2023-12-07 'fix small error in eval script at checkpoint loading'; 2024-02-08 visualisation fix; after that only README entries for new models. Server Last-Modified dates: task_D_D.zip 2023-02-23, task_ABC_D.zip and task_ABCD_D.zip 2022-09-15, debug zip 2022-05-13. The calvin_env submodule was last pushed 2024-01-03.",
     "short": "September 2025. Only a leaderboard entry was added."
    },
    "version": {
     "value": "No versioned releases",
     "display": "No tags or releases. Users run the main branch and the dataset currently on the server. The README changelog records breaking changes on 2022-01-10, 2022-02-07, 2022-09-16 and 2023-02-24.",
     "level": "verified",
     "sources": [
      "s8",
      "s9",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API: 0 releases, 0 tags.",
     "items": [
      {
       "value": "Three training/test splits",
       "display": "D→D (train and test in environment D), ABCD→D (train in all four, test in D) and ABC→D (train in A, B and C, test in the unseen environment D).",
       "level": "verified",
       "sources": [
        "s2",
        "s7"
       ]
      },
      {
       "value": "ABC→D is the most used split",
       "display": "The 2026 audit calls ABC→D CALVIN's most commonly used protocol.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      }
     ],
     "short": "No releases. Users run the main branch."
    },
    "capability": {
     "value": [
      "manipulation",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper targets long-horizon, language-conditioned manipulation: a policy must complete chains of five instructions in a row."
    },
    "generalisation": {
     "value": [
      "visual",
      "scene-layout",
      "language"
     ],
     "display": "ABC→D tests an unseen environment with different desk textures and moved drawer, sliding door, button and switch. Test instructions are phrasings not in the training set. D→D and ABCD→D test in an environment seen in training.",
     "level": "verified",
     "sources": [
      "s2",
      "s13",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Block start poses are not varied much at test time. The evaluation code places blocks at two fixed table positions per start condition, with a rotation drawn under a fixed seed; the audit says the released scene-D evaluation fixes block poses. The paper says all scene elements of D appear, in other positions, in the training environments.",
     "short": "New scene look and layout, and new wording"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All evaluation runs in simulation."
    },
    "simulator": {
     "value": "PyBullet",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The README FAQ says EGL GPU rendering is used for speed and that textures render slightly differently on GPU than on CPU.",
     "short": "PyBullet"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda (simulated, 7-DOF)",
     "level": "verified",
     "sources": [
      "s2",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "7-DOF arm with a parallel gripper whose fingers cannot be controlled independently. Control at 30 Hz."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "display": "A desk with a sliding door, a drawer, a button that toggles a green LED, a switch for a light bulb, and three coloured blocks.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 34,
     "display": "34 tasks, chained into sequences of 5",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT: the paper lists 34 tasks with success criteria (Fig. 6 and Fig. 9). The website leaderboard header says '(32 tasks)' for the MTLC column.",
     "short": "34 tasks"
    },
    "scenes": {
     "value": 4,
     "display": "4 environments (A, B, C, D) with different textures and different positions of the static elements",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "4 environments"
    },
    "objects": {
     "value": 3,
     "display": "3 movable blocks (red, blue, pink), plus a sliding door, a drawer, a button and a switch built into the desk",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper says objects were reduced to unicoloured primitive shapes on purpose.",
     "short": "3 blocks and 4 desk parts"
    },
    "demonstrations": {
     "value": "about 24 hours",
     "display": "About 24 hours of teleoperated play (about 6 hours per environment, about 2.4M interaction steps), collected by 3 untrained users with an HTC Vive VR headset. 1% of the data is labelled with language.",
     "level": "verified",
     "sources": [
      "s2",
      "s7",
      "s20",
      "s10"
     ],
     "checked": "2026-10-11",
     "note": "Play data has no fixed task list; users explored freely. Observations include static and gripper RGB-D cameras, tactile images and proprioception.",
     "items": [
      {
       "value": "Language labels",
       "display": "Labelled automatically by a task detector. The appendix gives 389 unique instructions for 34 tasks (about 11 per task); the main text says 'over 400'. The introduction says 20K language directives.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "CONFLICT inside the paper: 389 vs over 400 unique instructions."
      },
      {
       "value": "What '1%' means",
       "display": "The maintainers say 1% of 64-frame windows were labelled; training then cuts shorter sub-windows from them. A user counted about 40% of D→D training windows with language.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Download sizes",
       "display": "D→D 177,379,436,142 bytes; ABC→D 555,309,812,705 bytes; ABCD→D 704,022,347,117 bytes; debug 1,299,150,917 bytes. The README gives 166, 517 and 656 GB, which match these sizes in GiB.",
       "level": "verified",
       "sources": [
        "s10",
        "s7"
       ]
      },
      {
       "value": "Precomputed language embeddings",
       "display": "MiniLM embeddings ship with the data; since 2022-09-16 nine more embedding sets are on the server.",
       "level": "verified",
       "sources": [
        "s2",
        "s7"
       ]
      }
     ],
     "short": "About 24 hours of play data, with 1% labelled with language"
    },
    "scoring": {
     "value": [
      "chain-length",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "A subtask counts as solved when the simulator's task detector sees the required state change (for example a block lifted at least 5 cm)."
    },
    "metric_detail": {
     "value": "Average tasks completed in a row (Avg. Len., 0 to 5)",
     "display": "Main metric (LH-MTLC): a policy gets 1,000 fixed chains of 5 instructions and moves to the next instruction only if it solved the current one. Scores are the share of chains with 1, 2, 3, 4 and 5 tasks done in a row, and their sum, the average number completed (Avg. Len.). A second metric (MTLC) scores single tasks, but the README says it is only available for the baseline agent.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s11",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The original paper reports only the five in-a-row rates; the website and later papers add Avg. Len. By arithmetic, Avg. Len. equals the sum of the five rates.",
     "short": "Tasks done in a row, out of 5"
    },
    "trials": {
     "value": "1,000 fixed chains of 5 instructions",
     "level": "verified",
     "sources": [
      "s2",
      "s11",
      "s12",
      "s13",
      "s14"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Official manifest",
       "display": "1,000 chains generated with fixed random seeds from symbolic start states (drawer, slider, LED, light bulb and block locations). The robot is reset to a neutral pose before each chain.",
       "level": "verified",
       "sources": [
        "s2",
        "s12"
       ]
      },
      {
       "value": "Time limit",
       "display": "360 simulation steps per subtask (EP_LEN = 360), at 30 Hz.",
       "level": "verified",
       "sources": [
        "s11",
        "s2"
       ]
      },
      {
       "value": "One test sentence per subtask",
       "display": "The evaluation script uses the first entry of the validation annotation file for each subtask; the file holds one instruction per task (34 lines).",
       "level": "verified",
       "sources": [
        "s11",
        "s14"
       ]
      },
      {
       "value": "Seeds",
       "display": "Some papers average 3 training seeds (HULC, 3D Diffuser Actor, MDT, MoDE, FLOWER); others report one run. Seer reports the average of its top 3 checkpoints.",
       "level": "verified",
       "sources": [
        "s29",
        "s33",
        "s37",
        "s44",
        "s35"
       ]
      }
     ],
     "short": "1,000 chains of 5 tasks each"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s4",
      "s29",
      "s33",
      "s44",
      "s31",
      "s48",
      "s49"
     ],
     "checked": "2026-10-10",
     "note": "The website leaderboard shows point values only. HULC, 3D Diffuser Actor (v3), MDT, MoDE, GR-MG and FLOWER report standard deviations over 3 seeds. GR-1, RoboFlamingo, Seer, DreamVLA, X-VLA, Xiaomi-Robotics-0 and MMaDA-VLA report single numbers."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "verified",
     "sources": [
      "s5",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The README asks authors to contact Oier Mees to add a model. The leaderboard copies numbers from papers; no organiser re-runs submissions."
    },
    "leaderboard": {
     "value": "official",
     "display": "Maintainer-curated tables for D→D, ABCD→D and ABC→D on the project website. Last updated 2025-09-08.",
     "level": "verified",
     "sources": [
      "s4",
      "s5",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Not comprehensive. It lists 46 rows from papers up to FLOWER (2025-09). Higher published results (for example 4.78 on ABC→D and 4.80 on ABCD→D) are missing. Several rows differ from the source papers (see issues.i5).",
     "short": "Official and curated. Last updated in September 2025."
    },
    "top_score": {
     "value": 4.78,
     "display": "4.78 of 5 tasks completed in a row on average (ABC→D, MMaDA-VLA, March 2026). The highest ABC→D result without RL fine-tuning that we found. On ABCD→D the highest is 4.80 (Xiaomi-Robotics-0). Rows use different training data and are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s49",
      "s48",
      "s26",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Each score is verified at its source paper. 'Highest' is our judgement: it covers the 2026 audit's CALVIN tracker (snapshot 2026-05-21) and the papers we opened. After 2026-05 we found no higher ABC→D or ABCD→D score in two web searches on 2026-10-10; a full scan of arXiv after that date was not possible (the arXiv API returned errors and the shared search budget ran out). CHART NOTE: 'avg' here is Avg. Len. on a 0 to 5 scale, not a percentage; 'max' is 5 and 'chain' holds the reported 1 to 5 in-a-row rates in %.",
     "items": [
      {
       "value": 0.67,
       "display": "HULC, 2022-04: 41.8 / 16.5 / 5.7 / 1.9 / 1.1% in a row; Avg. Len. 0.67 (±0.1 over 3 seeds)",
       "level": "verified",
       "sources": [
        "s29"
       ],
       "data": {
        "model": "HULC",
        "date": "2022-04",
        "avg": 0.67,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         41.8,
         16.5,
         5.7,
         1.9,
         1.1
        ]
       }
      },
      {
       "value": 2.48,
       "display": "RoboFlamingo, 2023-11: Avg. Len. 2.48. Trained only on the language-labelled data.",
       "level": "verified",
       "sources": [
        "s30"
       ],
       "note": "The website lists 2.47 for the same row.",
       "data": {
        "model": "RoboFlamingo",
        "date": "2023-11",
        "avg": 2.48,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         82.4,
         61.9,
         46.6,
         33.1,
         23.5
        ]
       }
      },
      {
       "value": 3.06,
       "display": "GR-1 (ByteDance Research), 2023-12: Avg. Len. 3.06",
       "level": "verified",
       "sources": [
        "s31"
       ],
       "data": {
        "model": "GR-1",
        "date": "2023-12",
        "avg": 3.06,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         85.4,
         71.2,
         59.6,
         49.7,
         40.1
        ]
       }
      },
      {
       "value": 3.27,
       "display": "3D Diffuser Actor, 2024-02 (v1): Avg. Len. 3.27 with 60 keyposes",
       "level": "verified",
       "sources": [
        "s32",
        "s33"
       ],
       "note": "The same v1 also reports 3.83 with 360 keyposes; v3 (2024-07) reports 3.35 ± 0.04 over 3 seeds. The website uses 3.27.",
       "data": {
        "model": "3D Diffuser Actor",
        "date": "2024-02",
        "avg": 3.27,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         92.2,
         78.7,
         63.9,
         51.2,
         41.2
        ]
       }
      },
      {
       "value": 4.04,
       "display": "GR-MG, 2024-08: Avg. Len. 4.04 ± 0.03",
       "level": "verified",
       "sources": [
        "s34"
       ],
       "data": {
        "model": "GR-MG",
        "date": "2024-08",
        "avg": 4.04,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         96.8,
         89.3,
         81.5,
         72.7,
         64.4
        ]
       }
      },
      {
       "value": 4.28,
       "display": "Seer-Large, 2024-12: Avg. Len. 4.28 (average of the top 3 checkpoints)",
       "level": "verified",
       "sources": [
        "s35"
       ],
       "data": {
        "model": "Seer-Large",
        "date": "2024-12",
        "avg": 4.28,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         96.3,
         91.6,
         86.1,
         80.3,
         74
        ]
       }
      },
      {
       "value": 4.44,
       "display": "DreamVLA, 2025-07: Avg. Len. 4.44",
       "level": "verified",
       "sources": [
        "s43"
       ],
       "data": {
        "model": "DreamVLA",
        "date": "2025-07",
        "avg": 4.44,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         98.2,
         94.6,
         89.5,
         83.4,
         78.1
        ]
       }
      },
      {
       "value": 4.53,
       "display": "FLOWER, 2025-09: Avg. Len. 4.53 ± 0.04 (with pretraining)",
       "level": "verified",
       "sources": [
        "s44"
       ],
       "note": "Its five in-a-row rates sum to 4.49, not 4.53 (our arithmetic; see issues.i5).",
       "data": {
        "model": "FLOWER",
        "date": "2025-09",
        "avg": 4.53,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         99.4,
         95.8,
         90.7,
         84.9,
         77.8
        ]
       }
      },
      {
       "value": 4.43,
       "display": "X-VLA (0.9B), 2025-10: Avg. Len. 4.43",
       "level": "verified",
       "sources": [
        "s45"
       ],
       "data": {
        "model": "X-VLA (0.9B)",
        "date": "2025-10",
        "avg": 4.43,
        "rl": false,
        "split": "ABC→D",
        "max": 5
       }
      },
      {
       "value": 4.717,
       "display": "πRL on π0.5 (Flow-SDE), 2025-10: 4.717 after RL in scene D",
       "level": "verified",
       "sources": [
        "s46"
       ],
       "note": "Supervised training on the ABC data, then online RL with a per-subtask reward, scored in scene D. The paper calls this its in-distribution setting and says the RL ran under 'D→D training settings'. This is not a zero-shot ABC→D result.",
       "data": {
        "model": "πRL on π0.5 (Flow-SDE)",
        "date": "2025-10",
        "avg": 4.717,
        "rl": true,
        "split": "ABC data + RL in D",
        "max": 5,
        "chain": [
         99.7,
         98.2,
         95.8,
         91,
         87
        ]
       }
      },
      {
       "value": 4.75,
       "display": "Xiaomi-Robotics-0, 2026-02: Avg. Len. 4.75 on ABC→D (4.80 on ABCD→D)",
       "level": "verified",
       "sources": [
        "s48"
       ],
       "note": "Numbers read in v2 (2026-03-25).",
       "data": {
        "model": "Xiaomi-Robotics-0",
        "date": "2026-02",
        "avg": 4.75,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         100,
         98.3,
         96,
         92.6,
         88.1
        ]
       }
      },
      {
       "value": 4.78,
       "display": "MMaDA-VLA (8B), 2026-03: Avg. Len. 4.78",
       "level": "verified",
       "sources": [
        "s49",
        "s23"
       ],
       "note": "ACM Multimedia 2026. Size from the 2026 audit's Table 1.",
       "data": {
        "model": "MMaDA-VLA",
        "date": "2026-03",
        "avg": 4.78,
        "rl": false,
        "split": "ABC→D",
        "max": 5,
        "chain": [
         99.8,
         98.6,
         96.3,
         93.5,
         89.7
        ]
       }
      },
      {
       "value": "ABCD→D best",
       "display": "Xiaomi-Robotics-0, 2026-02: 4.80 (99.7 / 98.0 / 96.7 / 94.2 / 91.8%). Earlier: MDT-V 4.52 ± 0.02 (2024-07), GR-2 (ByteDance) 4.64 (2024-10), Being-H0.7 4.67 (2026-04).",
       "level": "verified",
       "sources": [
        "s48",
        "s37",
        "s38",
        "s50"
       ]
      },
      {
       "value": "D→D best found",
       "display": "FLOWER, 2025-09: 4.35 ± 0.02. MDT-V: 3.72 ± 0.05 (2024-07).",
       "level": "verified",
       "sources": [
        "s44",
        "s37"
       ],
       "note": "The audit's tracker lists a 4.48 D→D result from a 2026-04 paper (arXiv 2604.10170) that we did not check."
      },
      {
       "value": "Official leaderboard top rows",
       "display": "FLOWER on all three splits: ABC→D 4.53, ABCD→D 4.67, D→D 4.35",
       "level": "verified",
       "sources": [
        "s4"
       ]
      }
     ],
     "short": "4.78 of 5 (March 2026)",
     "chart": {
      "max": 5,
      "unit": "",
      "label": "Average number of tasks completed in a row, out of 5"
     }
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, Copyright (c) 2021 Oier Mees. The website says the code is 'for academic usage and is released under the MIT license'; the MIT text itself has no academic-only condition."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No licence for the dataset in the dataset README, download_data.sh, the main README or the website (all opened 2026-10-10). The dataset is served from calvin.cs.uni-freiburg.de. Third-party copies on Hugging Face carry self-assigned labels (items); they do not set CALVIN's own data licence.",
     "items": [
      {
       "value": "MIT",
       "display": "InternRobotics/InternData-Calvin_ABC (third-party copy, 3,883 downloads)",
       "level": "verified",
       "sources": [
        "s55"
       ]
      },
      {
       "value": "MIT",
       "display": "fywang/calvin-task-ABCD-D-lerobot (third-party copy)",
       "level": "verified",
       "sources": [
        "s55"
       ]
      },
      {
       "value": "Apache-2.0",
       "display": "ducido/calvin_task_D_D_* copies (third-party)",
       "level": "verified",
       "sources": [
        "s55"
       ]
      },
      {
       "value": "none",
       "display": "zhouhongyi/calvin_abc (third-party copy with no licence tag; 44,088 downloads)",
       "level": "verified",
       "sources": [
        "s55"
       ]
      }
     ],
     "short": "Not stated"
    },
    "license_assets": {
     "value": [
      "MIT",
      "Apache-2.0"
     ],
     "display": "The simulation assets (desk, blocks, Franka Panda model) sit in the calvin_env repository under its MIT licence; the Franka Panda folder carries its own Apache License 2.0 file.",
     "level": "inferred",
     "sources": [
      "s15",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Read by us from the calvin_env repository tree: the only licence files are the root MIT LICENSE and data/franka_panda/LICENSE.txt (Apache 2.0). The origin of the desk textures is not stated. The tactile simulator is a fork of facebookresearch/tacto (MIT). Not legal advice.",
     "short": "MIT, with the robot model under Apache-2.0"
    },
    "access": {
     "value": "open",
     "display": "Code on GitHub. Data from the University of Freiburg server over plain HTTP, with SHA-256 checksums. No registration.",
     "level": "verified",
     "sources": [
      "s7",
      "s10",
      "s21"
     ],
     "checked": "2026-10-11",
     "note": "HTTPS to calvin.cs.uni-freiburg.de is refused (checked 2026-10-10). Users in issues #103 and #116 (2025) report download speeds of tens of KB/s and broken unzips; some use parallel downloaders or proxies. Evaluating any split needs the scene-D data.",
     "short": "Open. HTTP download of 177 to 704 GB."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s6",
      "s15",
      "s16",
      "s7",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Code (MIT) and simulation assets (MIT, Apache-2.0) allow commercial use with attribution. No licence is stated for the dataset, so its terms are unknown. The website's phrase 'for academic usage' conflicts in tone with the MIT licence text. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Several policies scored on CALVIN were also run on real robots (for example GR-1, MDT, FLOWER), on different real tasks. No published study compares the same policies' CALVIN and real-robot scores.",
     "level": "inferred",
     "sources": [
      "s2",
      "s31",
      "s44",
      "s37",
      "s53",
      "s23",
      "s57"
     ],
     "checked": "2026-10-10",
     "note": "The paper says CALVIN captures challenges of real-world settings but has no real-robot experiment. GR-1, MDT and FLOWER report real-robot results separately from CALVIN. PolaRiS cites CALVIN among simulation benchmarks that 'sacrifice realism' and measures it no further. The 2026 audit calls a sim-vs-real ranking test impractical and does not run one. The basic entry graded this 'claimed'; we raise it to 'demonstrated' because the taxonomy defines that level as 'some policies also ran on real robots'.",
     "short": "No paired real-robot study"
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "CALVIN-Para",
      "CALVIN D (Enriched)",
      "CALVIN resampled-pose test",
      "GEVRM perturbed D"
     ],
     "display": "Variants built on CALVIN's scenes and tasks",
     "level": "verified",
     "sources": [
      "s51",
      "s30",
      "s23",
      "s28",
      "s52"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list. None has a published sim-to-real study that we found.",
     "items": [
      {
       "value": "CALVIN-Para",
       "display": "2026 (in the LIBERO-Para paper). 15 base tasks, 1,935 paraphrased instructions, single-task setting on ABCD→D weights, 5 seeds. See issues.i4.",
       "level": "verified",
       "sources": [
        "s51"
       ]
      },
      {
       "value": "CALVIN D (Enriched)",
       "display": "2023 (in the RoboFlamingo paper). Test instructions rewritten by GPT-4. See issues.i4.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "CALVIN resampled-pose and fresh-manifest tests",
       "display": "2026 (2026 audit). Block poses redrawn inside the training range, and two fresh 1,000-chain manifests. Code and per-sequence results released.",
       "level": "verified",
       "sources": [
        "s23",
        "s28"
       ]
      },
      {
       "value": "GEVRM perturbed D",
       "display": "2025 (ICLR 2025 method paper). Five perturbed versions of scene D; GR-1 falls to 1.44 average length. An ad hoc test, not a maintained benchmark.",
       "level": "verified",
       "sources": [
        "s52"
       ]
      }
     ],
     "short": "4 stress tests built on it"
    },
    "citations": {
     "value": 828,
     "display": "828 (Semantic Scholar; 132 influential)",
     "level": "verified",
     "sources": [
      "s22"
     ],
     "checked": "2026-10-10",
     "short": "828"
    },
    "github_stars": {
     "value": 996,
     "display": "996 stars, 128 forks (mees/calvin)",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "996"
    },
    "used_by": {
     "value": "At least 190 papers reported CALVIN results by 2026-05-21, by our count of the 2026 audit's CALVIN tracker (558 candidate papers). 21 of them have a first arXiv date in March 2026.",
     "level": "inferred",
     "sources": [
      "s26",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Counted by us from the audit's released tracker CSV: 190 rows classified 'reports-results', 328 'citation-only'. The audit calls its tracker counts lower bounds.",
     "items": [
      {
       "value": "GR-1, GR-MG, GR-2, RoboFlamingo, RoboVLMs",
       "display": "ByteDance Research and co-authors, 2023-2024",
       "level": "verified",
       "sources": [
        "s31",
        "s34",
        "s38",
        "s30",
        "s39"
       ]
      },
      {
       "value": "HULC, HULC++",
       "display": "University of Freiburg (CALVIN authors), 2022",
       "level": "verified",
       "sources": [
        "s29",
        "s5"
       ]
      },
      {
       "value": "MDT, MoDE, FLOWER",
       "display": "Karlsruhe Institute of Technology (FLOWER with Microsoft Research), 2024-2025",
       "level": "verified",
       "sources": [
        "s37",
        "s40",
        "s44"
       ]
      },
      {
       "value": "Seer",
       "display": "Shanghai AI Laboratory and Peking University, 2024-12",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "UniVLA (two different models share this name)",
       "display": "Bu et al. (HKU, OpenDriveLab, AgiBot), 3.80 on ABC→D; Wang et al., 4.41 on ABC→D and 4.63 on ABCD→D",
       "level": "verified",
       "sources": [
        "s42",
        "s41"
       ]
      },
      {
       "value": "X-VLA, VLA-Adapter, DreamVLA",
       "display": "2025",
       "level": "verified",
       "sources": [
        "s45",
        "s56",
        "s43"
       ]
      },
      {
       "value": "Xiaomi-Robotics-0",
       "display": "Xiaomi, 2026-02",
       "level": "verified",
       "sources": [
        "s48"
       ]
      },
      {
       "value": "MMaDA-VLA",
       "display": "Westlake University, Zhejiang University, Huawei and others, 2026-03",
       "level": "verified",
       "sources": [
        "s49"
       ]
      },
      {
       "value": "Being-H0.7",
       "display": "BeingBeyond, 2026-04",
       "level": "verified",
       "sources": [
        "s50"
       ]
      }
     ],
     "short": "At least 190 papers (May 2026)"
    },
    "industry_use": {
     "value": [
      "ByteDance",
      "Xiaomi",
      "BeingBeyond",
      "AgiBot",
      "Huawei",
      "Shanghai AI Laboratory"
     ],
     "level": "verified",
     "sources": [
      "s31",
      "s38",
      "s48",
      "s50",
      "s42",
      "s49",
      "s54"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ByteDance",
       "display": "GR-1, GR-MG and GR-2 report CALVIN results (GR-2: 4.64 on ABCD→D).",
       "level": "verified",
       "sources": [
        "s31",
        "s34",
        "s38"
       ]
      },
      {
       "value": "Xiaomi",
       "display": "Xiaomi-Robotics-0 reports 4.75 (ABC→D) and 4.80 (ABCD→D).",
       "level": "verified",
       "sources": [
        "s48"
       ]
      },
      {
       "value": "BeingBeyond",
       "display": "Being-H0.7 reports 4.48 (ABC→D) and 4.67 (ABCD→D).",
       "level": "verified",
       "sources": [
        "s50"
       ]
      },
      {
       "value": "AgiBot",
       "display": "Co-author of UniVLA (Bu et al.), which reports 3.80 on ABC→D.",
       "level": "verified",
       "sources": [
        "s42"
       ]
      },
      {
       "value": "Huawei",
       "display": "Co-author of MMaDA-VLA (4.78 on ABC→D).",
       "level": "verified",
       "sources": [
        "s49"
       ]
      },
      {
       "value": "Shanghai AI Laboratory (InternRobotics)",
       "display": "Seer reports CALVIN; the InternManip toolkit lists CALVIN as a supported benchmark and InternRobotics hosts a CALVIN ABC copy on Hugging Face.",
       "level": "verified",
       "sources": [
        "s35",
        "s54",
        "s55"
       ]
      }
     ]
    },
    "status": {
     "value": "dormant",
     "display": "No code change since 2024-02 and no leaderboard update since 2025-09-08. Use is very active.",
     "level": "inferred",
     "sources": [
      "s9",
      "s4",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "By the taxonomy rule (no updates for over a year): last commit and website change 2025-09-08, about 13 months before 2026-10-10. 52 open issues. The basic entry said 'maintained'.",
     "short": "No updates since September 2025. Use is very active."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "saturated",
     "title": "Top scores are close to the maximum of 5",
     "text": "On ABC→D, published averages reach 4.78 (MMaDA-VLA, 2026-03; 89.7% of chains fully completed), 4.75 (Xiaomi-Robotics-0) and 4.73 (VITA). On ABCD→D the best is 4.80 (Xiaomi-Robotics-0; 91.8% of chains fully completed). Several results since late 2025 lie within 0.1 of each other. The CALVIN-Para authors describe canonical CALVIN single-task performance as near-saturated (92.3% and 99.7% for two models).",
     "level": "verified",
     "sources": [
      "s49",
      "s48",
      "s47",
      "s51"
     ],
     "status": "open",
     "short": "The best average is 4.78 out of 5 on ABC→D, where policies train in environments A, B and C and are tested in D. On ABCD→D, which trains in all four, the best is 4.80."
    },
    {
     "id": "i2",
     "type": "shortcut",
     "title": "A small model given only a task ID scores like widely used policies",
     "text": "The 2026 audit trained a 0.09B probe (DINOv2 encoder and an MLP) that receives a 34-way task ID instead of the instruction. It scored 3.242 on ABC→D, 3.872 on ABCD→D and 3.123 on D→D (best checkpoint; final checkpoints 3.149 and 3.783). That is above RoboFlamingo on ABC→D (2.48) and close to it on ABCD→D (4.09), but well below the best results (4.78 and 4.80). The audit concludes the shortcut on CALVIN is real but smaller than on LIBERO. The official evaluation uses one fixed sentence per subtask, so the sentence identifies the subtask.",
     "level": "verified",
     "sources": [
      "s23",
      "s27",
      "s11",
      "s14"
     ],
     "status": "open",
     "note": "The audit's Table 1 lists MDT-V 4.52 as the best D→D result; the MDT paper reports 4.52 for ABCD→D and 3.72 for D→D (see issues.i5).",
     "short": "A small model given only a task number, without the instruction, scored 3.24 on ABC→D. That is above RoboFlamingo but below the best results."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Scores fall when block start poses are redrawn inside the training range",
     "text": "The released ABC→D evaluation places blocks at fixed poses, while during training blocks start anywhere in a range. The 2026 audit redrew block poses from that same range and kept everything else fixed (1,000 chains). Average tasks completed fell from 4.165 to 3.138 for X-VLA (95% CI of the drop 0.890 to 1.164), from 3.244 to 2.495 for GR-1 and from 2.367 to 1.869 for RoboFlamingo. Fully completed chains fell by 25.0, 13.5 and 6.4 points. Two fresh 1,000-chain manifests with the original pose rules moved scores by at most 0.11, within noise. The audit calls this distribution overfitting.",
     "level": "verified",
     "sources": [
      "s23",
      "s28",
      "s13"
     ],
     "status": "open",
     "short": "When the block start poses were redrawn inside the training range, the score of X-VLA fell from 4.17 to 3.14."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Policies fail when the instructions are reworded",
     "text": "CALVIN-Para paraphrases 15 base tasks into 1,935 instructions and tests them one task at a time with official ABCD→D weights. Success fell from 92.3% to 53.1% for RoboFlamingo and from 99.7% to 58.3% for FLOWER (5 seeds). The RoboFlamingo paper tested GPT-4-rewritten instructions on scene D: ABCD→D average length fell from 4.09 to 1.85 for RoboFlamingo and from 3.06 to 1.82 for HULC. CALVIN's own test uses one fixed sentence per subtask.",
     "level": "verified",
     "sources": [
      "s51",
      "s30",
      "s14"
     ],
     "status": "open",
     "short": "When the instructions were reworded, success on single tasks fell by about 40 points."
    },
    {
     "id": "i5",
     "type": "inconsistent-reporting",
     "title": "The same result appears with different numbers",
     "text": "3D Diffuser Actor reports 3.27 (v1, 60 keyposes), 3.83 (v1, 360 keyposes) and 3.35 ± 0.04 (v3) on ABC→D; the leaderboard uses 3.27. VPP reports 4.29 (v1) and 4.33 (v2). RoboFlamingo reports 2.48 and 4.09; the leaderboard shows 2.47 and 4.08. DeeR-VLA is 2.82 on the leaderboard and 2.90 in FLOWER's table. FLOWER's own rows do not add up: its five in-a-row rates sum to 4.49, 4.74 and 4.33, against reported averages of 4.53, 4.67 and 4.35 (our arithmetic); the leaderboard copies these rows. The leaderboard misses every result after 2025-09. The 2026 audit's Table 1 lists MDT-V 4.52 as the best D→D score, but MDT reports 4.52 for ABCD→D and 3.72 for D→D. Two different models are both called UniVLA (3.80 and 4.41 on ABC→D).",
     "level": "verified",
     "sources": [
      "s32",
      "s33",
      "s36",
      "s30",
      "s4",
      "s44",
      "s37",
      "s23",
      "s41",
      "s42"
     ],
     "status": "open",
     "note": "The sum checks are our arithmetic on the published tables (inferred).",
     "short": "The paper on 3D Diffuser Actor reports its score as 3.27, 3.35 and 3.83. Some leaderboard rows do not add up."
    },
    {
     "id": "i6",
     "type": "protocol-variance",
     "title": "Papers use different training recipes",
     "text": "Papers train on different data for the same split: all play data or only the language-labelled windows (3D Diffuser Actor's table marks this per method), with or without outside pretraining such as internet video, DROID or Open X-Embodiment. Some average 3 seeds, others report one run; Seer averages its top 3 checkpoints, and the benchmark has no separate validation set. πRL adds RL in scene D and reports the result next to ABC→D numbers. ABC→D, ABCD→D and D→D are often reported without saying which training data were used.",
     "level": "verified",
     "sources": [
      "s33",
      "s35",
      "s46",
      "s23",
      "s44",
      "s29"
     ],
     "status": "open",
     "short": "Papers differ in their training data, in how many random seeds they average and in how they choose checkpoints (saved versions of a model)."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Most claimed gains are not provably significant",
     "text": "The 2026 audit sorted 107 previous-best-to-new comparisons on ABC→D, using only public scores. 47 (43.9%) are provably significant at the 5% level, 36 (33.6%) cannot be decided from averages, 21 (19.6%) show no improvement and 3 (2.8%) are provably not significant. The released CSV later moves those 3 to 'no improvement'.",
     "level": "verified",
     "sources": [
      "s23",
      "s25",
      "s24"
     ],
     "status": "open",
     "note": "Percentages computed by us from the paper-version counts in the released CSV; they match the pie chart labels in the paper (Figure 3).",
     "short": "Of 107 claimed improvements on ABC→D, 44% can be shown to be statistically significant."
    },
    {
     "id": "i8",
     "type": "protocol-variance",
     "title": "Fixes without version numbers changed the data and the evaluation",
     "text": "The README changelog records a breaking change to evaluation start states (2022-01-10), changed success criteria for pushing and lifting (2022-02-07), wrong language annotations and scene files in the ABC and ABCD datasets fixed on 2022-09-16 (marked 'MAJOR BUG'), and a wrong scene file in the D dataset fixed on 2023-02-24. A bug in the LED button during rollouts was fixed in calvin_env on 2022-12-23 after issue #32. The repository has no version tags, so papers cannot state which version they used.",
     "level": "verified",
     "sources": [
      "s5",
      "s17",
      "s19",
      "s18",
      "s8"
     ],
     "status": "addressed",
     "note": "Results computed before these dates may have used the faulty data or code (inferred); we found no study of the size of the effect.",
     "short": "Bugs in the data and in the evaluation were fixed in 2022 and 2023 without version tags."
    },
    {
     "id": "i9",
     "type": "protocol-variance",
     "title": "Test runs differ across hardware",
     "text": "The README warns that GPU (EGL) rendering changes textures slightly compared with CPU rendering. The 2026 audit found that changing only the GPU (RTX A6000 vs RTX 6000 Ada) made GR-1's actions diverge from step 0. Changing only the CPU left 50 of 50 official sequences bitwise identical.",
     "level": "verified",
     "sources": [
      "s5",
      "s23"
     ],
     "status": "open",
     "short": "Changing only the GPU made the actions of GR-1 differ from the first step of a test run."
    },
    {
     "id": "i10",
     "type": "other",
     "title": "The dataset has no stated licence and one slow download source",
     "text": "No licence is stated for the CALVIN dataset. It is served from one university server over plain HTTP (ABC→D zip 555 GB). Users report downloads of tens of KB/s and broken unzips (issues #103 and #116, 2025-2026). Third-party copies on Hugging Face carry MIT, Apache-2.0, 'cc' or no licence labels.",
     "level": "verified",
     "sources": [
      "s7",
      "s4",
      "s10",
      "s21",
      "s55"
     ],
     "status": "open",
     "short": "The dataset has no stated licence. It is served from one HTTP server, and users report very slow downloads."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A high CALVIN score shows that a policy can chain trained tabletop skills at this simulated desk. On ABC→D, it also shows that the policy copes with a new desk look and layout. On its own, the score is weak evidence of general manipulation skill. A small model given only a task number matches common policies, redrawn block poses cut scores by up to one task, and reworded instructions cut success by about 40 points. No study ties CALVIN scores to real-robot results.",
     "basis": [
      "issues.i2",
      "issues.i3",
      "issues.i4",
      "facts.sim_to_real",
      "facts.generalisation"
     ],
     "confidence": "medium",
     "short": "A high score shows that a policy can chain trained skills at one simulated desk. It does not show general skill."
    },
    {
     "id": "r2",
     "text": "Treat gaps of about 0.1 in average length (the average number of tasks completed in a row) between top papers as ties. The same model is reported with differences of that size across versions of one paper, published rows do not always add up, and fewer than half of claimed ABC→D gains are provably significant.",
     "basis": [
      "issues.i1",
      "issues.i5",
      "issues.i7",
      "facts.top_score"
     ],
     "confidence": "high",
     "short": "Ignore gaps of about 0.1 between top results."
    },
    {
     "id": "r3",
     "text": "Always check which split (the combination of training and test environments) a CALVIN number comes from. ABC→D tests an unseen environment. ABCD→D and D→D test an environment seen in training, and ABCD→D scores run higher. Also check the training data, because outside pretraining and RL (reinforcement learning, which trains by trial and error) in the test scene are allowed and change scores.",
     "basis": [
      "facts.version",
      "facts.generalisation",
      "issues.i6"
     ],
     "confidence": "high",
     "short": "Check the split and the training data before comparing numbers."
    },
    {
     "id": "r4",
     "text": "The official leaderboard is a useful starting list, but it stops at September 2025 and differs from some source papers. Use the papers, or the 2026 audit's tracker, for current results.",
     "basis": [
      "facts.leaderboard",
      "facts.status",
      "issues.i5",
      "sources.s26"
     ],
     "confidence": "medium",
     "short": "The official leaderboard is out of date. Read the papers for current results."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy will do on a real robot.",
     "sub": "We found no study that scored the same policies on CALVIN and on real robots.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy copes with new start poses.",
     "sub": "When the block start poses were redrawn, scores fell by up to one task.",
     "basis": [
      "issues.i3"
     ]
    },
    {
     "id": "l3",
     "text": "Whether a policy understands new wording of an instruction.",
     "sub": "The test uses one fixed sentence for each subtask.",
     "basis": [
      "issues.i4",
      "facts.trials"
     ]
    },
    {
     "id": "l4",
     "text": "Whether a small gain over earlier work is real.",
     "sub": "Fewer than half of the claimed gains can be shown to be statistically significant.",
     "basis": [
      "issues.i7"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "CALVIN paper v4 (no real-robot experiments); README and website; GR-1, MDT and FLOWER papers (real-robot results reported separately); PolaRiS (2512.16881: cites CALVIN as sacrificing realism, no measurement); 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (2606.10366: CALVIN in related work only); X2Real (2609.27449: related work only); 2026 audit (calls the test impractical). One web search on 2026-10-10 for CALVIN sim-real correlation found no paired study.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data",
     "where": "dataset/README.md, download_data.sh, main README, LICENSE, website, calvin_env repository tree, dataset server directory (HTTP 403).",
     "date": "2026-10-10"
    },
    {
     "for": "top_score (newer than 2026-05)",
     "where": "2026 audit tracker (snapshot 2026-05-21); two web searches on 2026-10-10 for CALVIN ABC→D averages above 4.78; arXiv API listing attempts on 2026-10-10 returned HTTP 503; the shared web-search budget then ran out. Not re-searched after that.",
     "date": "2026-10-10"
    },
    {
     "for": "derived_benchmarks",
     "where": "One web search for CALVIN-based variants (2026-10-10), the LIBERO-Para, RoboFlamingo, GEVRM and audit papers.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard comprehensiveness",
     "where": "Website tables (46 rows) compared with the audit tracker and the source papers listed under top_score.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks (arXiv abstract page, submission history v1 to v4)",
     "url": "https://arxiv.org/abs/2112.03227",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2021-12",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "CALVIN paper, full text v4 (RA-L accepted version)",
     "url": "https://arxiv.org/pdf/2112.03227v4",
     "type": "paper",
     "publisher": "arXiv / IEEE Robotics and Automation Letters",
     "date": "2022-07",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Crossref record for DOI 10.1109/LRA.2022.3180108 (RA-L vol. 7, no. 3, pp. 7327-7334)",
     "url": "https://api.crossref.org/works/10.1109/lra.2022.3180108",
     "type": "index",
     "publisher": "Crossref",
     "date": "2022-07",
     "accessed": "2026-10-11"
    },
    "s4": {
     "title": "CALVIN project website and leaderboard (served over HTTP only; HTTPS refused)",
     "url": "http://calvin.cs.uni-freiburg.de/",
     "type": "leaderboard",
     "publisher": "Autonomous Intelligent Systems Lab, University of Freiburg",
     "date": "2025-09-08",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "CALVIN GitHub README (evaluation, FAQ, changelog, model list)",
     "url": "https://github.com/mees/calvin/blob/main/README.md",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "CALVIN LICENSE file",
     "url": "https://github.com/mees/calvin/blob/main/LICENSE",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "CALVIN dataset README (splits, sizes, data structure)",
     "url": "https://github.com/mees/calvin/blob/main/dataset/README.md",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "GitHub API: mees/calvin (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/mees/calvin",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "CALVIN commit history (releases and tags lists are empty)",
     "url": "https://github.com/mees/calvin/commits/main",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2025-09-08",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "CALVIN dataset server: sha256sum.txt and HTTP headers of the four zip files (size, Last-Modified)",
     "url": "http://calvin.cs.uni-freiburg.de/dataset/sha256sum.txt",
     "type": "repo",
     "publisher": "University of Freiburg",
     "date": "2023-02-24",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "CALVIN evaluation script evaluate_policy.py (EP_LEN = 360, NUM_SEQUENCES = 1000, first validation instruction per subtask)",
     "url": "https://github.com/mees/calvin/blob/main/calvin_models/calvin_agent/evaluation/evaluate_policy.py",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "CALVIN sequence generator multistep_sequences.py (fixed seeds)",
     "url": "https://github.com/mees/calvin/blob/main/calvin_models/calvin_agent/evaluation/multistep_sequences.py",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "CALVIN evaluation utils.py (get_env_state_for_initial_condition: fixed block positions)",
     "url": "https://github.com/mees/calvin/blob/main/calvin_models/calvin_agent/evaluation/utils.py",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "CALVIN test instructions new_playtable_validation.yaml (one instruction per task)",
     "url": "https://github.com/mees/calvin/blob/main/calvin_models/conf/annotations/new_playtable_validation.yaml",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "calvin_env LICENSE (MIT) and repository tree (table, block and robot assets)",
     "url": "https://github.com/mees/calvin_env/blob/main/LICENSE",
     "type": "repo",
     "publisher": "mees/calvin_env",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "calvin_env data/franka_panda/LICENSE.txt (Apache License 2.0)",
     "url": "https://github.com/mees/calvin_env/blob/main/data/franka_panda/LICENSE.txt",
     "type": "repo",
     "publisher": "mees/calvin_env",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "calvin_env commit 797142c: fix bug in button during rollouts",
     "url": "https://github.com/mees/calvin_env/commit/797142c588c21e76717268b7b430958dbd13bf48",
     "type": "repo",
     "publisher": "mees/calvin_env",
     "date": "2022-12-23",
     "accessed": "2026-10-11"
    },
    "s18": {
     "title": "CALVIN issue #40: some inconsistencies in the dataset",
     "url": "https://github.com/mees/calvin/issues/40",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2023-02",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "CALVIN issue #32: Major concern about evaluation",
     "url": "https://github.com/mees/calvin/issues/32",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022-12",
     "accessed": "2026-10-11"
    },
    "s20": {
     "title": "CALVIN issue #23: proportion of data with language instructions",
     "url": "https://github.com/mees/calvin/issues/23",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2022-08",
     "accessed": "2026-10-11"
    },
    "s21": {
     "title": "CALVIN issues #103 and #116: slow dataset downloads and broken unzips",
     "url": "https://github.com/mees/calvin/issues/116",
     "type": "repo",
     "publisher": "mees/calvin",
     "date": "2025-08",
     "accessed": "2026-10-11"
    },
    "s22": {
     "title": "Semantic Scholar batch API record for ARXIV:2112.03227",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (2026 audit; full text v1)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Manipulation Benchmark Audit project page",
     "url": "https://ripl.github.io/manipulation_benchmark_audit/",
     "type": "site",
     "publisher": "TTIC RIPL (lists CoRL 2026 and IROS 2026 RGMCW Workshop)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Audit release: statistical_significance_pie_counts.csv and statistical_significance_bucket_comparison.csv",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/analysis/release_current_values/statistical_significance_pie_counts.csv",
     "type": "repo",
     "publisher": "ripl/ManipulationBenchmarkAudit",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Audit CALVIN citation tracker (calvin_citation_tracker.csv, snapshot 2026-05-21)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/leaderboards/calvin/calvin_citation_tracker.csv",
     "type": "repo",
     "publisher": "ripl/ManipulationBenchmarkAudit",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Audit CALVIN probe results (shortcut_solvability/results/calvin summaries)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/tree/main/shortcut_solvability/results/calvin",
     "type": "repo",
     "publisher": "ripl/ManipulationBenchmarkAudit",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Audit CALVIN resampled-pose config and distribution_overfitting_summary.csv",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/creeping_overfitting/results/calvin/distribution_overfitting_summary.csv",
     "type": "repo",
     "publisher": "ripl/ManipulationBenchmarkAudit",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "What Matters in Language Conditioned Robotic Imitation Learning over Unstructured Data (HULC)",
     "url": "https://arxiv.org/abs/2204.06252",
     "type": "paper",
     "publisher": "IEEE RA-L (University of Freiburg)",
     "date": "2022-04",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Vision-Language Foundation Models as Effective Robot Imitators (RoboFlamingo)",
     "url": "https://arxiv.org/abs/2311.01378",
     "type": "paper",
     "publisher": "arXiv (ByteDance Research, Tsinghua, SJTU, NUS)",
     "date": "2023-11",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation (GR-1)",
     "url": "https://arxiv.org/abs/2312.13139",
     "type": "paper",
     "publisher": "arXiv (ByteDance Research)",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "3D Diffuser Actor: Policy Diffusion with 3D Scene Representations, v1",
     "url": "https://arxiv.org/pdf/2402.10885v1",
     "type": "paper",
     "publisher": "arXiv (Carnegie Mellon University)",
     "date": "2024-02",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "3D Diffuser Actor, v3 (current version)",
     "url": "https://arxiv.org/pdf/2402.10885",
     "type": "paper",
     "publisher": "arXiv (Carnegie Mellon University)",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "GR-MG: Leveraging Partially-Annotated Data via Multi-Modal Goal-Conditioned Policy",
     "url": "https://arxiv.org/abs/2408.14368",
     "type": "paper",
     "publisher": "IEEE RA-L",
     "date": "2024-08",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Predictive Inverse Dynamics Models are Scalable Learners for Robotic Manipulation (Seer)",
     "url": "https://arxiv.org/abs/2412.15109",
     "type": "paper",
     "publisher": "arXiv (Shanghai AI Laboratory, Peking University, CUHK)",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Video Prediction Policy (VPP), v1 and v2",
     "url": "https://arxiv.org/abs/2412.14803",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Multimodal Diffusion Transformer: Learning Versatile Behavior from Multimodal Goals (MDT)",
     "url": "https://arxiv.org/abs/2407.05996",
     "type": "paper",
     "publisher": "arXiv (Karlsruhe Institute of Technology)",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation",
     "url": "https://arxiv.org/abs/2410.06158",
     "type": "paper",
     "publisher": "arXiv (ByteDance Research)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "What Matters in Building Vision-Language-Action Models for Generalist Robots (RoboVLMs)",
     "url": "https://arxiv.org/abs/2412.14058",
     "type": "paper",
     "publisher": "arXiv (Tsinghua, ByteDance Research and others)",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "Efficient Diffusion Transformer Policies with Mixture of Expert Denoisers for Multitask Learning (MoDE)",
     "url": "https://arxiv.org/abs/2412.12953",
     "type": "paper",
     "publisher": "arXiv (Karlsruhe Institute of Technology)",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "Unified Vision-Language-Action Model (UniVLA, Wang et al.)",
     "url": "https://arxiv.org/abs/2506.19850",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "Learning to Act Anywhere with Task-centric Latent Actions (UniVLA, Bu et al.)",
     "url": "https://arxiv.org/abs/2505.06111",
     "type": "paper",
     "publisher": "arXiv (The University of Hong Kong, OpenDriveLab, AgiBot)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge",
     "url": "https://arxiv.org/abs/2507.04447",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "FLOWER: Democratizing Generalist Robot Policies with Efficient Vision-Language-Action Flow Policies",
     "url": "https://arxiv.org/abs/2509.04996",
     "type": "paper",
     "publisher": "arXiv (Karlsruhe Institute of Technology, Microsoft Research)",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "X-VLA: Soft-Prompted Transformer as Scalable Cross-Embodiment Vision-Language-Action Model",
     "url": "https://arxiv.org/abs/2510.10274",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "πRL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models (v3, Appendix C.3 and D.2)",
     "url": "https://arxiv.org/abs/2510.25889",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "Unifying Perception and Action: A Hybrid-Modality Pipeline with Implicit Visual Chain-of-Thought (VITA)",
     "url": "https://arxiv.org/abs/2511.19859",
     "type": "paper",
     "publisher": "arXiv (Nanjing University)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "Xiaomi-Robotics-0: An Open-Sourced Vision-Language-Action Model with Real-Time Execution (v2)",
     "url": "https://arxiv.org/abs/2602.12684",
     "type": "paper",
     "publisher": "Xiaomi Robotics",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "MMaDA-VLA: Large Diffusion Vision-Language-Action Model with Unified Multi-Modal Instruction and Generation (v3)",
     "url": "https://arxiv.org/abs/2603.25406",
     "type": "paper",
     "publisher": "ACM Multimedia 2026 (Westlake University, Zhejiang University, Huawei and others)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "Being-H0.7: A Latent World-Action Model from Egocentric Videos",
     "url": "https://arxiv.org/abs/2605.00078",
     "type": "paper",
     "publisher": "BeingBeyond",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s51": {
     "title": "LIBERO-Para: A Diagnostic Benchmark and Metrics for Paraphrase Robustness in VLA Models (v3, Appendix E: CALVIN-Para)",
     "url": "https://arxiv.org/abs/2603.28301",
     "type": "paper",
     "publisher": "arXiv (EMNLP 2026 per arXiv comment)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s52": {
     "title": "GEVRM: Goal-Expressive Video Generation Model for Robust Visual Manipulation (perturbed CALVIN D)",
     "url": "https://arxiv.org/abs/2502.09268",
     "type": "paper",
     "publisher": "ICLR 2025 (Zhejiang University, Westlake University)",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s53": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies",
     "url": "https://arxiv.org/abs/2512.16881",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s54": {
     "title": "InternManip README (benchmarks supported)",
     "url": "https://github.com/InternRobotics/InternManip",
     "type": "repo",
     "publisher": "InternRobotics",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s55": {
     "title": "Hugging Face Hub dataset listing for 'calvin' (licence tags of third-party copies)",
     "url": "https://huggingface.co/api/datasets?search=calvin&limit=100&full=true",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s56": {
     "title": "VLA-Adapter: An Effective Paradigm for Tiny-Scale Vision-Language-Action Model",
     "url": "https://arxiv.org/abs/2509.09372",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s57": {
     "title": "A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation",
     "url": "https://arxiv.org/abs/2606.10366",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from the checked basic entry and primary sources. Corrections to the basic entry: status set to dormant (no update for over a year), sim_to_real raised from 'claimed' to 'demonstrated', top score updated from the leaderboard (4.53) to the literature (4.78 ABC→D, 4.80 ABCD→D), leaderboard rows found inconsistent with source papers. Checking continued into 2026-10-11 (local time); sources opened then carry that access date."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "causalvqa",
   "name": "CausalVQA",
   "aliases": [],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Video-QA on physical cause and effect in human activity clips from Ego-Exo4D; no robot. Taxonomy-v0 marks egocentric human video benchmarks 'borderline'.",
   "summary": {
    "text": "Video question set on cause and effect in real Ego-Exo4D activity clips: counterfactual, anticipation, planning.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "All six authors: FAIR at Meta",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-06 (arXiv v1 2025-06-11)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "Last repo commit 2025-08-18 'fixed quality of life functions for submission'.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "version": {
     "value": "arXiv v1 only",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Searched OpenReview API (term 'CausalVQA') and Semantic Scholar (venue arXiv.org)."
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Counterfactual, hypothetical, anticipation, planning, descriptive questions about physical events."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "1,586 items (793 paired questions), 779 video segments (Sec. 2.3.1 and datasheet).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "Humans 84.78%; best model Gemini 2.5 Flash 61.66% paired.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Official (: Meta 'Physical Reasoning from Video' HF Space (held-out split, 793 question pairs, answers private). README link facebook/pwm_leaderboard redirects there. Space showed 'Runtime error' on 2026-10-10.)"
    },
    "license_data": {
     "value": "Benchmark available under the Ego-Exo4D licence; data fetched from the Ego4D consortium S3 bucket after sign-up and approval (up to 48 h).",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "license_code": {
     "value": "LICENSE.pdf in repo is the Ego-Exo4D licence agreement draft (licensor Carnegie Mellon University); no separate code licence. Model scripts adapted from lmms-eval.",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "GitHub reports no licence."
    },
    "access": {
     "value": "application",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Not built to predict robot performance. Searched paper, repo."
    },
    "citations": {
     "value": 38,
     "display": "38 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 62,
     "display": "62",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "used_by": {
     "value": "NVIDIA Cosmos 3 report (video, physical and causal reasoning evaluation).",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Introduction and human-evaluation section say 1,786 items; composition section and datashe",
     "text": "Internal inconsistency in arXiv v1; not resolved.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models",
     "url": "https://arxiv.org/pdf/2506.09943v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s2": {
     "title": "CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models",
     "url": "https://arxiv.org/abs/2506.09943",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s3": {
     "title": "facebookresearch/CausalVQA on GitHub (repository)",
     "url": "https://github.com/facebookresearch/CausalVQA",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "facebook/physical_reasoning_leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/facebook/physical_reasoning_leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "facebookresearch/CausalVQA on GitHub (file README.md)",
     "url": "https://github.com/facebookresearch/CausalVQA/blob/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "facebookresearch/CausalVQA on GitHub (file LICENSE.pdf)",
     "url": "https://github.com/facebookresearch/CausalVQA/blob/main/LICENSE.pdf",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s8": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2606.02800v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "chores",
   "name": "CHORES",
   "full_name": "CHORES (SPOC)",
   "aliases": [
    "Chores",
    "CHORES-S",
    "CHORES-L",
    "ChoresNav"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores a simulated Stretch robot on household navigation and manipulation tasks in procedurally generated houses.",
   "summary": {
    "text": "Ten-task simulated household benchmark where a Stretch robot finds, picks up and fetches objects in generated houses.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "14 authors incl. Kiana Ehsani, Tanmay Gupta, Rose Hendrix, Jordi Salvador, Luca Weihs, Aniruddha Kembhavi; repo allenai/spoc-robot-training",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2023-12 (arXiv v1 2023-12-05)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "arXiv v2 2024-08-07; repo last pushed 2024-11-04",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "CVPR 2024 (CVF Open Access page)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Fetched via curl; page title 'CVPR 2024 Open Access Repository'. Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "mobile-manipulation",
      "navigation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "ObjectNav, PickUp, Fetch, RoomVisit; tasks given as templated natural-language specs (repo README). Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "robots": {
     "value": "Hello Robot Stretch RE-1, RGB cameras only",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scale": {
     "value": "10-task suite; ~200,000 procedurally generated houses; ~40,000 unique 3D assets (abstract) / 41,133 (paper body); average 195 evaluation episodes per task",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Success rate, episode-length-weighted success (SEL), % rooms visited. Checked 2026-10-10."
    },
    "top_score": {
     "value": "Paper: SPOC multitask 49.9% avg success on CHORES-S (65.0% with ground-truth detection); 38.6% on CHORES-L",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: no leaderboard found)",
     "note": "Looked at project page and repo README. Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Tried, not measured (: 2 models (SPOC RGB-only, SPOC w/ Detic), 88 real trials in two physical settings. Real averages 39.5% and 56.1%. ObjectNav 55.0% sim vs 50.0% real. No correlation statistic.)",
     "note": "Two models run on a real Stretch RE-1 in two physical settings, 88 trials; e.g. ObjectNav 55.0% sim vs 50.0% real. No correlation statistic."
    },
    "citations": {
     "value": 72,
     "display": "72",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 157,
     "display": "157",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Last push 2024-11. Checked 2026-10-10."
    },
    "version": {
     "value": "No releases or tags; paper v2 (2024-08-07)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "license_data": {
     "value": "Training data and houses distributed via the Apache-2.0 repo scripts; Objaverse-derived assets not checked",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World",
     "url": "https://arxiv.org/abs/2312.02976",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-12"
    },
    "s2": {
     "title": "allenai/spoc-robot-training on GitHub (repository)",
     "url": "https://github.com/allenai/spoc-robot-training",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "CVPR 2024 Open Access Repository",
     "url": "https://openaccess.thecvf.com/content/CVPR2024/html/Ehsani_SPOC_Imitating_Shortest_Paths_in_Simulation_Enables_Effective_Navigation_and_CVPR_2024_paper.html",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World (full text)",
     "url": "https://arxiv.org/html/2312.02976v2",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-12"
    },
    "s5": {
     "title": "SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World",
     "url": "https://spoc-robot.github.io/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "allenai/spoc-robot-training on GitHub (blob)",
     "url": "https://github.com/allenai/spoc-robot-training/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2312.02976",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "cosmos-humaneval",
   "name": "Cosmos-HumanEval",
   "full_name": "Cosmos-HumanEval (Cosmos-HUE)",
   "aliases": [
    "Cosmos HUE",
    "HUE-PaiBench v1.2",
    "Cosmos Human-Eval",
    "nvidia/Cosmos-HumanEval-v1"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Human evaluation of generated video for physical AI. Robot prompts are 17 of 100 I2V and 17 of 97 T2V samples in the open set; other domains are human, driving, physics, industry, common, misc. General video-physics evaluation, so 'borderline' under taxonomy-v0. Built by NVIDIA for its robot-oriented Cosmos world models.",
   "summary": {
    "text": "NVIDIA human-rating protocol: annotators answer yes/no physics and fidelity questions about generated videos.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "NVIDIA Corporation (dataset owner); introduced in the Cosmos 3 report by NVIDIA.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2026-05",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "HF dataset created 2026-05-05 (API); card lists 'Dataset Creation Date: 2026-05-20'. Cosmos 3 report v1 2026-06-01."
    },
    "latest_update": {
     "value": "HF dataset lastModified 2026-06-09; Cosmos 3 report v4 submitted 2026-06-23.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "HF date from API."
    },
    "version": {
     "value": "v1.2-opensource ('HUE-PaiBench v1.2'); the card calls it the publicly releasable subset of NVIDIA's HUE question bank.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Humans rate generated videos; no control loop."
    },
    "capability": {
     "value": [
      "world-modeling"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Four axes: Visual Integrity, Semantic Alignment, Physical Laws, Geometric Reasoning."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "197 samples (100 I2V, 97 T2V); 2,957 questions; per category: Physical Laws 822, Visual Integrity 771, Semantic Alignment 754, Geometric Reasoning 610.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "T2V evaluation pool: 100 prompts sampled from PAIBench-G, 5 seeds each, up to 20 questions per video, up to 10,000 binary observations per checkpoint.",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "note": "Same appendix also says each video receives 'up to 16' questions; the card says 14-16 per sample. Flagged as an internal inconsistency."
      }
     ]
    },
    "scoring": {
     "value": [
      "human-rating"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "evaluator": {
     "value": "organiser-run",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "The organisers (NVIDIA annotators). The card also allows VLM-as-judge use of the same questions.)"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: T2V and I2V leaderboards in Cosmos 3 report Appendix F (Tables 32-33) and Table 14.)"
    },
    "top_score": {
     "value": "T2V: Veo-3.1 91.3, Seedance-1.5-Pro 90.0, Cosmos3-Super 89.3; real-video ground truth 93.6. I2V: Veo-3.1 89.7, Cosmos3-Super 89.6; ground truth 94.4.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Scored by the model's own developer (NVIDIA)."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Dataset holds only two JSON question files; no scoring code found on the card. Cosmos 3 abstract says code, checkpoints and an evaluation benchmark are released under OpenMDW-1.1, but we did not find HUE scoring code."
    },
    "license_data": {
     "value": "custom: OpenMDW-1.1; card adds that the data was created in part with GPT-5.2 and may not be used to develop or train AI/ML systems. Card also says it is ready for commercial or non-commercial use.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No evidence that HUE scores predict robot policy success. The report validates HUE by real-video ground truth (93.6 T2V, 94.4 I2V) and binomial confidence intervals, not by robot outcomes. Looked in Cosmos 3 report and dataset card."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Report says question-bank refinement is ongoing."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2606.02800v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    },
    "s2": {
     "title": "nvidia/Cosmos-HumanEval-v1 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/nvidia/Cosmos-HumanEval-v1",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s3": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI",
     "url": "https://arxiv.org/abs/2606.02800",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "dialfred",
   "name": "DialFRED",
   "aliases": [
    "Dialogue-Enabled Agents for Embodied Instruction Following",
    "DialFRED Challenge"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores a simulated household agent (ALFRED/AI2-THOR) that may ask questions while completing tasks; fixed splits, metric and a challenge leaderboard.",
   "summary": {
    "text": "ALFRED extension where the agent may ask a human questions about locations, appearance, or direction to finish tasks.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "UCLA (Xiaofeng Gao, Ran Gong); Amazon Alexa AI (Qiaozi Gao, Kaixiang Lin, Govind Thattai, Gaurav Sukhatme); USC (Sukhatme). Work supported by Amazon Alexa AI.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "UCLA lead author; industry co-authors."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2022-02 (arXiv v1 2022-02-27)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "EvalAI DialFRED Challenge 2022-10-01 to 2023-06-30 (phase 'CVPR 2023 EAI Workshop'); repo last commit 2023-09-14",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Commit date via GitHub API."
    },
    "version": {
     "value": "arXiv v2 (2022-08-15)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "instruction-following",
      "collaboration",
      "navigation",
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "53K human QA pairs for 29,376 sub-goals; 25 sub-goal types; 3 question types; 112 rooms; 80 object types; ~$10K annotation cost",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "Paper best (RL anytime questioner): val seen SR 47.8, val unseen SR 33.6, unseen PWSR 20.4; instruction-only baseline unseen SR 18.3. Challenge board: host baseline SR 0.303, Team Keio SR 0.14 (2 public entries).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Challenge numbers from https://eval.ai/api/jobs/challenge_phase_split/4368/leaderboard/"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Official (EvalAI DialFRED Challenge, new test set; 2 public entries)"
    },
    "license_code": {
     "value": "Creative Commons Attribution-NonCommercial 4.0 International (repo LICENSE file); THIRD_PARTY folder carries ALFRED's MIT licence",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API reports NOASSERTION because CC BY-NC is not a recognised code licence."
    },
    "license_data": {
     "value": "CC BY-NC 4.0 (human QA CSV ships inside the repo under the repo LICENSE)",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Data file ./data/dialfred_human_qa.csv is in the same repo; no separate data licence found."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No real-robot evaluation in the paper or repo; none found in a web search."
    },
    "citations": {
     "value": 110,
     "display": "110 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 47,
     "display": "47 (xfgao/DialFRED)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Challenge closed 2023-06; last commit 2023-09."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "DialFRED: Dialogue-Enabled Agents for Embodied Instruction Following",
     "url": "https://arxiv.org/abs/2202.13330",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-02"
    },
    "s2": {
     "title": "DialFRED: Dialogue-Enabled Agents for Embodied Instruction Following",
     "url": "https://arxiv.org/pdf/2202.13330v2",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-02"
    },
    "s3": {
     "title": "https://eval.ai/api/challenges/challenge/1859/",
     "url": "https://eval.ai/api/challenges/challenge/1859/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "EvalAI: Evaluating state of the art in AI",
     "url": "https://eval.ai/web/challenges/challenge-page/1859/overview",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "xfgao/DialFRED on GitHub (blob)",
     "url": "https://github.com/xfgao/DialFRED/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/ARXIV:2202.13330?fields=citationCount",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s7": {
     "title": "xfgao/DialFRED on GitHub (repository)",
     "url": "https://api.github.com/repos/xfgao/DialFRED",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "xfgao/DialFRED on GitHub (repository)",
     "url": "https://github.com/xfgao/DialFRED",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "droid",
   "name": "DROID",
   "full_name": "DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset",
   "aliases": [
    "Distributed Robot Interaction Dataset",
    "DROID platform",
    "droid_101",
    "r2d2_faceblur"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "A training dataset collected on one shared hardware platform (Franka arm, standard cameras and software). That platform is the base of RoboArena's distributed real-robot evaluation, and DROID-trained policies are tested zero-shot in new scenes. Several simulated and world-model evaluations (PolaRiS, REALM, Ctrl-World, RoboWorld) have been checked against real DROID-platform results.",
   "summary": {
    "text": "DROID is a set of 76k successful teleoperated demonstrations (350 hours) on a Franka arm, recorded in 564 real scenes by 13 institutions using identical hardware. It has no fixed test, but its hardware platform and DROID-trained policies are the basis of RoboArena and of the most-studied real-to-sim evaluations.",
    "short": "DROID is a dataset of 76k real robot demonstrations recorded in 564 scenes on one shared robot setup. It has no fixed test. Labs use the same setup to test policies (the models that control robots).",
    "sources": [
     "s2",
     "s4",
     "s21"
    ]
   },
   "facts": {
    "kind": {
     "value": "dataset",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Released as a dataset with training code, checkpoints and a hardware guide. No fixed test or scoring rule."
    },
    "kind_secondary": {
     "value": [
      "study"
     ],
     "display": "Also a shared hardware platform, and a one-off evaluation in the paper (6 tasks, 4 locations)",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The taxonomy's 'platform' means a simulator, so the hardware platform role has no value."
    },
    "publishers": {
     "value": [
      "Stanford University",
      "UC Berkeley",
      "Toyota Research Institute",
      "13 collecting institutions"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s1"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Authors",
       "display": "101 authors; first authors Alexander Khazatsky and Karl Pertsch. 16 numbered affiliations including Stanford, UC Berkeley, Toyota Research Institute, CMU, UT Austin, Princeton, University of Washington, Google DeepMind and KAIST.",
       "level": "verified",
       "sources": [
        "s1",
        "s2"
       ]
      },
      {
       "value": "Count of collecting groups differs",
       "display": "'13 institutions' and '18 robots' (Sections I and III) vs '18 research labs' (introduction). Raw data folders are split by 13 lab names.",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "consortium",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Academic-led multi-institution effort with Toyota Research Institute and Google DeepMind."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Lead institutions are in North America; collection spanned North America, Asia and Europe."
    },
    "first_release": {
     "value": "2024-03",
     "display": "arXiv v1 2024-03-19; RLDS 1.0.0 files in the bucket dated 2024-03-15. Published at RSS 2024.",
     "level": "verified",
     "sources": [
      "s1",
      "s10",
      "s3"
     ],
     "checked": "2026-10-10",
     "short": "March 2024, at RSS 2024"
    },
    "published_at": {
     "value": "RSS 2024",
     "display": "Robotics: Science and Systems XX, paper 120, July 2024, DOI 10.15607/RSS.2024.XX.120",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2025-09",
     "display": "2025-09-01: updated idle-frame filter in the official annotations repo. 2025-09-15: last repo commit (docs and site config). RLDS 1.0.1 objects in the bucket carry 2025-07-15 timestamps.",
     "level": "verified",
     "sources": [
      "s13",
      "s7",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Site news: December 2024 language annotations; April 2025 improved calibrations for 36k episodes; arXiv v2 2025-04-22. A Hugging Face port named droid_1.0.1 existed from 2025-03-17, so version 1.0.1 predates its bucket timestamps.",
     "short": "September 2025. A data filter was updated."
    },
    "version": {
     "value": "RLDS 1.0.1",
     "display": "RLDS 1.0.0 (92,233 episodes) and 1.0.1 (95,658 episodes); raw 1.0.0 and 1.0.1 (face-blurred stereo video); droid_100 sample (100 episodes). Extra annotations ship separately on Hugging Face.",
     "level": "verified",
     "sources": [
      "s10",
      "s11",
      "s9",
      "s5",
      "s13"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "1.0.0 vs 1.0.1 language labels",
       "display": "openpi: 1.0.1 has the complete language annotations (about 75k episodes), 1.0.0 only 30k. The official annotations repo says the released RLDS dataset contains only a subset of labels.",
       "level": "verified",
       "sources": [
        "s15",
        "s13"
       ],
       "note": "The 30k figure is Physical Intelligence's; the official repo gives no number for 1.0.0."
      },
      {
       "value": "OXE copy",
       "display": "OXE v1.1 lists DROID with 92,233 episodes (1,670 GB), the 1.0.0 count.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      }
     ],
     "short": "RLDS 1.0.1"
    },
    "status": {
     "value": "dormant",
     "display": "No official update since September 2025. Maintainers answered issues in 2025. Use as training data and test platform is very active.",
     "level": "inferred",
     "sources": [
      "s7",
      "s13",
      "s6",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "54 open issues and PRs; users still file data questions in 2026 (e.g. #80 on redistributing derived labels, #73 stereo availability). Promised scene-type metadata (2024-04) has not appeared in a dataset version.",
     "short": "No official updates since September 2025. Use is very active."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "generalisation": {
     "value": [
      "object-pose",
      "object-instance",
      "visual",
      "scene-layout"
     ],
     "display": "Paper: noisy start positions, distractors, new objects, a camera shift. Later: zero-shot use in unseen scenes (FAST, RoboArena).",
     "level": "verified",
     "sources": [
      "s2",
      "s20",
      "s21"
     ],
     "checked": "2026-10-10",
     "short": "New positions, objects, visuals and scenes"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "none",
     "display": "None. Simulated copies of the platform exist (PolaRiS, REALM).",
     "level": "verified",
     "sources": [
      "s2",
      "s23",
      "s25"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Panda",
     "display": "Franka Panda 7-DoF arm with Robotiq 2F-85 gripper on a movable standing desk; two Zed 2 stereo cameras and a wrist Zed Mini; Oculus Quest 2 teleoperation; 15 Hz",
     "level": "verified",
     "sources": [
      "s2",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The hardware list puts the total cost at about $46,000 (updated around September 2025; earlier $20,000), mostly the arm.",
     "short": "Franka Panda"
    },
    "scene": {
     "value": [
      "home",
      "office-lab",
      "kitchen",
      "mixed"
     ],
     "display": "Offices, kitchens, bedrooms, bathrooms, living rooms and labs across 52 buildings; scene types assigned by GPT-4V",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper says 10 scene types (Section IV) and 9 (Fig. 10 caption)."
    },
    "tasks": {
     "value": 86,
     "display": "86 tasks, counted as unique verbs in the instructions (paper body, site, RSS abstract). The arXiv abstract says 84.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT between the arXiv abstract and the body. A verb count is not comparable with task counts of fixed benchmarks.",
     "short": "86 tasks, counted as verbs"
    },
    "scenes": {
     "value": 564,
     "display": "564 scenes in 52 buildings; 1,417 camera viewpoints",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "564 scenes"
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The paper plots object categories (extracted from instructions) without a total. Not on the site or docs."
    },
    "demonstrations": {
     "value": 76000,
     "display": "76k successful demonstrations (350 hours) by 50 collectors over 12 months. About 16k trajectories marked 'not successful' are also released. The RSS abstract says 65k.",
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT: 76k (arXiv, site) vs 65k (RSS 2024 abstract). RLDS episode counts include failures.",
     "items": [
      {
       "value": "RLDS episode counts",
       "display": "1.0.0: 92,233 episodes (1,834,749,018,029 bytes). 1.0.1: 95,658 episodes (1,865,993,126,270 bytes).",
       "level": "inferred",
       "sources": [
        "s10",
        "s11"
       ],
       "note": "Summed by us from shardLengths in each dataset_info.json."
      },
      {
       "value": "Up to 3 crowdsourced instructions per episode",
       "display": "Labelled after collection on the tasq.ai platform; 3 labels for 95% of the 75k successful episodes since December 2024.",
       "level": "verified",
       "sources": [
        "s2",
        "s4"
       ]
      }
     ],
     "short": "76k demonstrations, 350 hours"
    },
    "scale": {
     "value": "1.7 TB RLDS",
     "display": "RLDS 1.7 TB; raw stereo HD video 8.7 TB; raw non-stereo 5.6 TB (docs)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "A user reports the non-stereo raw subset is 3.9 TB (issue #60, open).",
     "short": "1.7 TB (RLDS)"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Later DROID-platform evaluations use progress scores (PolaRiS, REALM) or pairwise preferences (RoboArena)."
    },
    "metric_detail": {
     "value": "Paper: A/B success rates of co-trained policies",
     "display": "Paper: diffusion policies trained on small in-domain data, with or without 50/50 co-training on DROID or OXE; 6 tasks at 4 locations, in-distribution and out-of-distribution. DROID beat the next best by 22 points (in distribution) and 17 points (OOD).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Subset used",
       "display": "The paper's policies used the first 40K successful trajectories that had language labels at the time.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "FAST 'DROID evaluation'",
       "display": "First zero-shot test of DROID policies in unseen scenes (2025-01): 44 trials per policy over 17 tasks in its Table II (its text says 16).",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "RoboArena",
       "display": "Pairwise blind A/B tests of DROID policies by evaluators at many sites; ratings from a task-aware Bradley-Terry model. Has its own record.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Offline action error",
       "display": "NVIDIA's GR00T DROID example reports open-loop action error on DROID episodes (average MSE about 0.0149) as expected performance.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ],
     "short": "Success rate. Later tests use other scores."
    },
    "trials": {
     "value": "10 per task setting (paper); varies later",
     "display": "Paper: 10 rollouts per task setting and method. FAST: 44 per policy. RoboArena paper: 4,284 episodes. PolaRiS: 20 real rollouts per policy and environment.",
     "level": "verified",
     "sources": [
      "s2",
      "s20",
      "s21",
      "s23"
     ],
     "checked": "2026-10-10",
     "short": "10 per task setting"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s20",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "The paper reports standard errors; FAST reports 95% intervals; RoboArena shows a standard deviation per rating. Many later reports give single numbers."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Papers run their own DROID-platform trials. RoboArena, a separate project, collects crowd-sourced evaluations of submitted policies."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the DROID site. RoboArena keeps a live DROID-platform leaderboard (24 policies on 2026-10-11 UTC)."
    },
    "license_code": {
     "value": "none (droid repo); MIT (droid_policy_learning)",
     "display": "The hardware and control code repo has no LICENSE file (GitHub licence API: null). The policy-learning repo is MIT (copyright 2021 Stanford Vision and Learning Lab, a robomimic fork).",
     "level": "verified",
     "sources": [
      "s6",
      "s8",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "In issue #62 (2025-07) a user asked for a licence; a maintainer pointed to the data licence in the bucket. No licence was added to the repo.",
     "short": "None for the control code. MIT for the policy code."
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "display": "CC BY 4.0 (paper; licence file in the bucket next to RLDS 1.0.0)",
     "level": "verified",
     "sources": [
      "s2",
      "s12",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "There is no licence file in the 1.0.1 folder. Hugging Face ports carry a different label (items); they do not change the original terms.",
     "items": [
      {
       "value": "apache-2.0",
       "display": "lerobot/droid_1.0.1 (Hugging Face; 22,682 downloads) and cadene/droid_1.0.1 (163,135 downloads), Hub 'downloads' field",
       "level": "verified",
       "sources": [
        "s35",
        "s36"
       ],
       "note": "Read from the dataset cards' metadata."
      }
     ],
     "short": "CC BY 4.0"
    },
    "license_assets": {
     "value": "not applicable",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Real-robot recordings only. People's faces in the raw video were blurred before release (docs; internal name r2d2_faceblur)."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s5",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Public bucket gs://gresearch/robotics/droid via TFDS or gsutil; no registration.",
     "short": "Open, in a public storage bucket"
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s12",
      "s6",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The data (CC BY 4.0) and policy code (MIT) allow commercial use with attribution. The hardware and control software repo has no licence, so reuse rights for it are not granted. The dataset alone reads as allowed. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "display": "Scores on DROID hardware come from real robots. Simulated and world-model copies of the platform have been checked against real DROID results (see validity).",
     "level": "inferred",
     "sources": [
      "s23",
      "s24",
      "s25",
      "s26",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "Eight paired comparisons found. Physics simulators: PolaRiS r 0.90 (authors overlap); REALM r 0.92 (independent); an independent re-test found REALM r 0.785, a SIMPLER-style setup r 0.402 and VLA-Arena r 0.725 on real DROID hardware. World models trained on DROID: Ctrl-World's own data gives r 0.83 by our calculation, but PolaRiS measured Ctrl-World at r 0.53; RoboWorld reports r 0.989 against the RoboArena leaderboard. Offline action error: r -0.55 and -0.53 (PolaRiS). All use 3 to 8 policies.",
     "short": "Scores come from real robots. Many simulated and learned copies have been checked against them."
    },
    "real_reproducibility": {
     "value": "multi-site-measured",
     "display": "RoboArena ran DROID-platform evaluations at 7 universities and compared the rankings with an exhaustive 'oracle' ranking.",
     "level": "inferred",
     "sources": [
      "s21",
      "s2",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "RoboArena (authors overlap with DROID): 7 policies, 612 pairwise comparisons, 4,284 episodes. Under simulated distribution shifts, its ranking matched the oracle at r 0.838 (MMRV 0.058) vs 0.692 (0.141) for a single-lab standard test. The DROID paper tested at 4 locations with different tasks; FAST showed one checkpoint at 3 campuses without measuring success."
    },
    "citations": {
     "value": 1183,
     "display": "1,183 (Semantic Scholar; 105 influential)",
     "level": "verified",
     "sources": [
      "s34"
     ],
     "checked": "2026-10-10",
     "short": "1,183"
    },
    "github_stars": {
     "value": 449,
     "display": "droid 449 stars (100 forks); droid_policy_learning 305 stars (30 forks)",
     "level": "verified",
     "sources": [
      "s6",
      "s38"
     ],
     "checked": "2026-10-10",
     "short": "449"
    },
    "dataset_downloads": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The official copy is a Google Cloud bucket with no public download counter. Hugging Face ports: cadene/droid_1.0.1 163,135 downloads (4,968,753 all time), lerobot/droid_1.0.1 22,682 (182,981 all time), Hub API 2026-10-10."
    },
    "used_by": {
     "value": "At least 9 model reports train on DROID or test on its platform",
     "level": "verified",
     "sources": [
      "s20",
      "s14",
      "s21",
      "s28",
      "s29",
      "s30",
      "s31",
      "s32",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "A lower bound from reports we opened.",
     "items": [
      {
       "value": "OpenVLA",
       "display": "2024-06. Franka-DROID fine-tuning tests. DROID was in pretraining at 10% weight but removed for the last third because action-token accuracy stayed low.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "π0-FAST-DROID, π0-DROID, π0.5-DROID",
       "display": "Physical Intelligence, 2025. Checkpoints in openpi; FAST paper reports the first zero-shot DROID tests in unseen scenes.",
       "level": "verified",
       "sources": [
        "s20",
        "s14"
       ]
      },
      {
       "value": "SpatialVLA",
       "display": "2025-01. Removed DROID for the final third of pretraining, following OpenVLA.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "GR00T N1, N1.7",
       "display": "NVIDIA. DROID in N1 pretraining (23.1M frames, 428.3 h); GR00T-N1.7-DROID checkpoint.",
       "level": "verified",
       "sources": [
        "s30",
        "s31"
       ]
      },
      {
       "value": "V-JEPA 2-AC",
       "display": "Meta, 2025-06. Action-conditioned model trained on 23k DROID trajectories, including failures.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "RoboArena entrants",
       "display": "24 policies on the live leaderboard, e.g. pi05_droid (1608 ± 30.7, 745 evaluations), allenai/MolmoAct2-DROID, Cosmos3-Nano-Policy, G0.5-Droid-AR.",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "note": "Leaderboard API, last updated 2026-10-11 03:30 UTC."
      }
     ],
     "short": "At least 9 model reports"
    },
    "industry_use": {
     "value": [
      "Physical Intelligence",
      "NVIDIA",
      "Meta",
      "Toyota Research Institute",
      "Hugging Face"
     ],
     "level": "verified",
     "sources": [
      "s14",
      "s31",
      "s32",
      "s2",
      "s35"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Physical Intelligence",
       "display": "Three DROID checkpoints, a DROID training guide and idle filter in openpi; π0.5-DROID is on RoboArena.",
       "level": "verified",
       "sources": [
        "s14",
        "s15"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "DROID in GR00T N1 pretraining; GR00T-N1.7-DROID checkpoint and example.",
       "level": "verified",
       "sources": [
        "s30",
        "s31"
       ]
      },
      {
       "value": "Meta",
       "display": "V-JEPA 2-AC trained on DROID.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "Toyota Research Institute",
       "display": "One of the lead affiliations; a raw data folder (TRI) holds its collection.",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      },
      {
       "value": "Hugging Face",
       "display": "Hosts the LeRobot port lerobot/droid_1.0.1.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "RoboArena",
      "PolaRiS",
      "REALM",
      "Ctrl-World",
      "RoboWorld",
      "RobotArena ∞ DROIDSim"
     ],
     "display": "Evaluations built on the DROID platform or data",
     "level": "verified",
     "sources": [
      "s21",
      "s23",
      "s25",
      "s24",
      "s27",
      "s33"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RoboArena",
       "display": "2025-06. Distributed real-robot A/B evaluation on DROID hardware. Has its own record.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "PolaRiS",
       "display": "2025-12. Simulated scenes rebuilt from video scans (Gaussian splats, Isaac Sim).",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "REALM",
       "display": "2025-12. Isaac Sim benchmark with 15 perturbation factors; tasks from DROID episodes.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Ctrl-World, RoboWorld",
       "display": "2025-10 and 2026-07. Video world models trained on DROID and used to score policies.",
       "level": "verified",
       "sources": [
        "s24",
        "s27"
       ]
      },
      {
       "value": "RobotArena ∞ DROIDSim",
       "display": "2025-10. Simulated scenes from DROID videos; no real correlation measured.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      }
     ],
     "short": "6 evaluations built on it"
    }
   },
   "validity": [
    {
     "id": "v1",
     "name": "PolaRiS (simulated DROID scenes)",
     "date": "2025-12",
     "by": "authors",
     "method": "The same 5 DROID-trained policies were scored in 6 reconstructed scenes and on real robots at UW and Princeton, with 20 real rollouts (test runs) per policy and scene.",
     "result": "Pearson correlation r = 0.90 and MMRV (a measure of how often two rankings disagree) = 0.03. The worst scene had r = 0.81.",
     "authors_view": "strong correlation",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s23"
     ],
     "note": "Figures 6, 7, 13. Policies are co-fine-tuned 1k steps on about 350 simulated demonstrations first. 6 of 14 PolaRiS authors are DROID authors."
    },
    {
     "id": "v2",
     "name": "PolaRiS compared with RoboArena",
     "date": "2025-12",
     "by": "authors",
     "method": "The PolaRiS scores of 4 policies were compared with their average progress scores on RoboArena.",
     "result": "Pearson r = 0.98 and MMRV = 0.00",
     "authors_view": "strong performance correlation",
     "n_policies": 4,
     "level": "verified",
     "sources": [
      "s23"
     ],
     "note": "Figure 8."
    },
    {
     "id": "v3",
     "name": "Offline action error (PolaRiS baseline)",
     "date": "2025-12",
     "by": "authors",
     "method": "The same 5 policies were ranked by action error on the training set and on the validation set, and the rankings were compared with real performance.",
     "result": "Pearson r = -0.55 on the training set and -0.53 on the validation set. MMRV = 0.40.",
     "authors_view": "poor metric",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s23"
     ],
     "note": "Figure 7 labels. The paper does not name the data; the policies were trained on DROID, so this is most likely DROID data (our inference)."
    },
    {
     "id": "v4",
     "name": "Ctrl-World (world model, own study)",
     "date": "2025-10",
     "by": "authors",
     "method": "DROID checkpoints of π0, π0-FAST and π0.5 were run on 7 tasks from the same start images, in a DROID-trained world model (a model that predicts what the cameras will see next) and on a real DROID setup.",
     "result": "No coefficient was reported. The regression slopes were 0.87 for instruction following and 0.81 for success. By our calculation, Pearson r = 0.97 and 0.83 over 21 task-policy pairs.",
     "authors_view": "closely correlated",
     "n_policies": 3,
     "level": "inferred",
     "sources": [
      "s24"
     ],
     "note": "Computed by us from Table 3. The world model underestimates success (policy means 0.343 / 0.479 / 0.757 real vs 0.193 / 0.293 / 0.486). Overlap with DROID authors: Finn."
    },
    {
     "id": "v5",
     "name": "Ctrl-World as tested by PolaRiS",
     "date": "2025-12",
     "by": "authors",
     "method": "The same 5 policies were run in Ctrl-World across the 6 scenes of PolaRiS, scored by humans and compared with real results.",
     "result": "Pearson r = 0.53 and MMRV = 0.22",
     "authors_view": "clear policy mis-rankings",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s23"
     ]
    },
    {
     "id": "v6",
     "name": "REALM (Isaac Sim, own study)",
     "date": "2025-12",
     "by": "independent",
     "method": "GR00T N1.5, π0 and π0-FAST ran 7 tasks under 5 perturbations in simulation and on a real DROID setup. There were about 800 paired rollouts, scored by task progress.",
     "result": "Overall Pearson r = 0.92, MMRV = 0.118 and p < 0.001",
     "authors_view": "strong proxy",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s25"
     ],
     "note": "Fig. 6 (PDF): default setting r 0.88, MMRV 0.166; camera-pose perturbation r 0.89, MMRV 0.253. Correlation is over task-policy-perturbation points."
    },
    {
     "id": "v7",
     "name": "Independent re-test of REALM, SIMPLER and VLA-Arena",
     "date": "2026-06",
     "by": "independent",
     "method": "5 policies (π0, π0-FAST, π0.5, GR00T N1.6 and GR00T N1.7) ran 9 matched tasks in three simulators and on real DROID hardware. There were 11,800 simulated and 1,115 real rollouts.",
     "result": "REALM had a mean Spearman of 0.700, a mean Pearson of 0.785 and a mean MMRV of 0.030. SIMPLER had 0.400, 0.402 and 0.128. VLA-Arena had 0.575, 0.725 and 0.060.",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s26"
     ],
     "note": "Table 2. Only 5 real rollouts per policy, task and perturbation dimension. Fine-tuning in REALM raised its Spearman correlation to 0.875."
    },
    {
     "id": "v8",
     "name": "RoboWorld (world model) compared with RoboArena",
     "date": "2026-07",
     "by": "independent",
     "method": "8 open policies were run in a DROID-trained world model from RoboArena start frames, with 4,186 rollouts. The results were compared with the RoboArena leaderboard.",
     "result": "Pearson r = 0.989 and Spearman = 0.970",
     "authors_view": "align strongly",
     "n_policies": 8,
     "level": "verified",
     "sources": [
      "s27"
     ],
     "note": "Leaderboard snapshot 2026-02-26; data dump 2026-02-03."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How one study's result compares with another study's result.",
     "sub": "There is no fixed test. Each study uses its own.",
     "basis": [
      "facts.metric_detail",
      "issues.i5"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a low error on recorded data means success on a real robot.",
     "sub": "Action error (how far a policy's actions are from the recorded ones) correlated negatively with real results.",
     "basis": [
      "issues.i6"
     ]
    },
    {
     "id": "l3",
     "text": "How a policy would do on other kinds of robots.",
     "sub": "All of the data comes from one Franka setup.",
     "basis": [
      "facts.robots",
      "facts.embodiment"
     ]
    }
   ],
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Counts differ between sources and versions",
     "text": "Tasks: 86 (paper body, site, RSS) vs 84 (arXiv abstract). Trajectories: 76k (arXiv, site) vs 65k (RSS abstract). Episodes: 92,233 (RLDS 1.0.0) vs 95,658 (1.0.1); Ctrl-World cites 95,599. Scene types: 10 (Section IV) vs 9 (Fig. 10). Collecting groups: 13 institutions vs 18 labs.",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s3",
      "s10",
      "s11",
      "s24"
     ],
     "status": "open",
     "short": "The counts of tasks, trajectories and episodes differ between sources."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "Data fixes were released after the dataset",
     "text": "The official annotations repo says the original release 'included noisy camera calibrations' (better ones for about 36k episodes since April 2025), that the released RLDS contains only a subset of the language labels (full labels since December 2024), and that many episodes contain long pauses, mostly at the start, which make policies output idle actions. It recommends filtering them with a list valid only for version 1.0.1 (added August-September 2025). Physical Intelligence's openpi guide says idle filtering significantly improves policy performance.",
     "level": "verified",
     "sources": [
      "s13",
      "s15",
      "s4"
     ],
     "status": "addressed",
     "mitigation": {
      "text": "Fixes ship as separate files on Hugging Face (KarlP/droid), not as a new RLDS version.",
      "sources": [
       "s13"
      ]
     },
     "short": "Camera calibrations, language labels and idle frames (pauses where the robot does not move) needed fixes after the release."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Some models could not fit DROID in large training mixtures",
     "text": "OpenVLA trained with DROID at 10% weight but removed it for the last third of training because action-token accuracy stayed low; SpatialVLA did the same. The FAST paper says OpenVLA struggled to fit the higher-frequency DROID data and that FAST enabled the first strong generalist DROID policy. RobotArena ∞ states DROID is often left out of pretraining because of higher noise, citing a third paper.",
     "level": "verified",
     "sources": [
      "s28",
      "s29",
      "s20",
      "s33"
     ],
     "status": "open",
     "note": "The RobotArena ∞ statement is a claim about other people's practice; we did not check it.",
     "short": "Some models dropped DROID from their training mix because they could not fit it."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Some raw video is missing after face blurring",
     "text": "Since 2024-03-19 the docs have said 20% of raw episodes were missed when face-blurring and copying the raw data, 'to be fixed within a few days'. The note is still there. In 2025-03 a maintainer said a few episodes were likely missed in the extra face-blurring step for the non-RLDS data. A raw 1.0.1 folder dated 2024-03-18 exists, but the docs do not say whether it is complete. The RLDS version is complete per the docs.",
     "level": "verified",
     "sources": [
      "s5",
      "s18",
      "s17",
      "s9"
     ],
     "status": "open",
     "short": "Some raw video episodes are missing. A 2024 note about this was never resolved."
    },
    {
     "id": "i5",
     "type": "protocol-variance",
     "title": "There is no standard test on DROID hardware",
     "text": "Results come from different protocols: the paper's co-training A/B tests (6 tasks), FAST's 44-trial suite, RoboArena's crowd-sourced pairwise ratings, and simulated proxies scored by task progress. RoboArena found that a conventional single-lab suite (FAST's) ranked policies less accurately than distributed evaluation (r 0.692 vs 0.838 with an exhaustive ranking, under simulated shifts).",
     "level": "verified",
     "sources": [
      "s2",
      "s20",
      "s21"
     ],
     "status": "open",
     "short": "DROID results come from test protocols that are not compatible with each other."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Offline action error is a poor predictor of real performance",
     "text": "In PolaRiS, action error on training and validation data correlated negatively with real performance of 5 DROID policies (r -0.55 and -0.53). NVIDIA's GR00T DROID example still reports open-loop action error as its expected performance figure.",
     "level": "verified",
     "sources": [
      "s23",
      "s31"
     ],
     "status": "open",
     "short": "Lower action error did not mean better real performance."
    },
    {
     "id": "i7",
     "type": "protocol-variance",
     "title": "Proxy evaluations disagree with each other",
     "text": "World models: Ctrl-World's own data gives r 0.83 (our calculation), but PolaRiS measured it at r 0.53 with clear mis-rankings; RoboWorld reports r 0.989 on RoboArena. Simulators: REALM reported r 0.92 on its own; an independent group measured r 0.785 for REALM and 0.402 for a SIMPLER-style setup on DROID hardware. Each study used 3 to 8 policies.",
     "level": "verified",
     "sources": [
      "s24",
      "s23",
      "s27",
      "s25",
      "s26"
     ],
     "status": "contested",
     "counter": {
      "text": "PolaRiS, REALM and RoboWorld each report strong agreement within their own setups (r 0.90, 0.92, 0.989).",
      "sources": [
       "s23",
       "s25",
       "s27"
      ],
      "short": "Each proxy's authors report strong agreement in their own setup."
     },
     "short": "A proxy (a simulated or learned stand-in for real tests) scores well in its own paper. It scores worse when other groups test it."
    },
    {
     "id": "i8",
     "type": "other",
     "title": "Licences are missing for the code and later data folders",
     "text": "The hardware and control code repo has no licence file; a maintainer pointed users to the data licence in the bucket (2025-07). The CC BY 4.0 file sits only in the 1.0.0 folder. Hugging Face ports label the data Apache-2.0. A user asked in 2026-09 whether derived labels may be redistributed; no answer yet.",
     "level": "verified",
     "sources": [
      "s6",
      "s16",
      "s12",
      "s35",
      "s39"
     ],
     "status": "open",
     "short": "The control code has no licence. Copies of the data carry a different licence label."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "DROID matters for evaluation as a shared robot setup more than as a dataset. Because many labs own the same hardware, policies trained on it can be compared in the real world (RoboArena) and in calibrated simulations. A DROID-platform number is only meaningful with its protocol named.",
     "basis": [
      "facts.derived_benchmarks",
      "issues.i5",
      "facts.real_reproducibility"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "DROID matters as a shared robot setup more than as a test. Always name the protocol behind a DROID number."
    },
    {
     "id": "r2",
     "text": "DROID has the most paired evidence comparing simulation and real robots of any setup we have checked. The studies are small (3 to 8 policies) and mostly by overlapping teams, and they disagree when one group tests another's proxy. Treat any single proxy score as a rough indication.",
     "basis": [
      "facts.sim_to_real",
      "issues.i7"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Many studies have checked proxies against real robots. The studies are small and they conflict."
    },
    {
     "id": "r3",
     "text": "Before training on DROID, use version 1.0.1 with the separate annotation, calibration and idle-filter files. Results from older pipelines are not directly comparable.",
     "basis": [
      "issues.i2",
      "facts.version"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Use version 1.0.1 together with the fix files released later."
    }
   ],
   "searched": [
    {
     "for": "validity / sim_to_real",
     "where": "DROID paper v2 (no sim); PolaRiS v2 (Figures 6-8, 13); Ctrl-World v3 (Table 3, PDF Figure 7); REALM v1 (PDF Fig. 6); 2606.10366 (Table 2); RoboWorld v4; RobotArena ∞ (no real correlation); RoboArena v2 (real vs oracle, not sim); 2508.11117 (position paper, no measurement); WorldEval and dWorldEval (do not use DROID).",
     "date": "2026-10-10"
    },
    {
     "for": "objects",
     "where": "Paper text and figure captions, site, docs: no total.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code, licence files",
     "where": "droid repo root and GitHub licence API; droid_policy_learning LICENSE; bucket folders droid/1.0.0, droid/1.0.1, droid_raw; issue #62.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "No headline score for the dataset. RoboArena ratings belong to the RoboArena record.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2403.12945",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "DROID paper, full text v2",
     "url": "https://arxiv.org/html/2403.12945v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "DROID, RSS 2024 proceedings page (paper 120)",
     "url": "https://www.roboticsproceedings.org/rss20/p120.html",
     "type": "paper",
     "publisher": "Robotics: Science and Systems",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "DROID project site (updates, Hugging Face link)",
     "url": "https://droid-dataset.github.io/",
     "type": "site",
     "publisher": "DROID team",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "DROID docs: The DROID Dataset (download sizes, schema, raw-data note)",
     "url": "https://droid-dataset.github.io/droid/the-droid-dataset",
     "type": "site",
     "publisher": "DROID team",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: droid-dataset/droid (licence null, stars, issues)",
     "url": "https://api.github.com/repos/droid-dataset/droid",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "droid repo commit history",
     "url": "https://github.com/droid-dataset/droid/commits/main",
     "type": "repo",
     "publisher": "DROID team",
     "date": "2025-09-15",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "droid_policy_learning LICENSE (MIT)",
     "url": "https://github.com/droid-dataset/droid_policy_learning/blob/master/LICENSE",
     "type": "repo",
     "publisher": "DROID team",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Google Cloud Storage listing: gs://gresearch/robotics/droid, droid_raw, droid_100",
     "url": "https://storage.googleapis.com/storage/v1/b/gresearch/o?prefix=robotics/droid&delimiter=/",
     "type": "dataset",
     "publisher": "Google (bucket)",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "DROID RLDS 1.0.0 dataset_info.json",
     "url": "https://storage.googleapis.com/gresearch/robotics/droid/1.0.0/dataset_info.json",
     "type": "dataset",
     "publisher": "DROID team",
     "date": "2024-03-15",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "DROID RLDS 1.0.1 dataset_info.json",
     "url": "https://storage.googleapis.com/gresearch/robotics/droid/1.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "DROID team",
     "date": "2025-07-15",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "CC-BY-4.0 licence file in the DROID 1.0.0 bucket folder",
     "url": "https://storage.googleapis.com/gresearch/robotics/droid/1.0.0/CC-BY-4.0",
     "type": "dataset",
     "publisher": "DROID team",
     "date": "2024-03-15",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "KarlP/droid on Hugging Face: official annotations, calibrations, idle filter (README and commits)",
     "url": "https://huggingface.co/KarlP/droid",
     "type": "repo",
     "publisher": "Karl Pertsch (DROID co-first author)",
     "date": "2025-09-01",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "openpi README (DROID checkpoints)",
     "url": "https://github.com/Physical-Intelligence/openpi",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "openpi DROID training guide (version 1.0.1, idle filtering)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/examples/droid/README_train.md",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "droid issue #62: License (maintainer reply)",
     "url": "https://github.com/droid-dataset/droid/issues/62",
     "type": "repo",
     "publisher": "DROID team (issue tracker)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "droid issue #47: No corresponding raw data for many RLDS episodes (maintainer reply)",
     "url": "https://github.com/droid-dataset/droid/issues/47",
     "type": "repo",
     "publisher": "DROID team (issue tracker)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "History of docs/the-droid-dataset.md (raw-data note added 2024-03-19)",
     "url": "https://github.com/droid-dataset/droid/commits/main/docs/the-droid-dataset.md",
     "type": "repo",
     "publisher": "DROID team",
     "date": "2025-03-11",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "DROID docs: hardware shopping list (approximate total cost)",
     "url": "https://github.com/droid-dataset/droid/blob/main/docs/hardware-setup/shopping-list.md",
     "type": "repo",
     "publisher": "DROID team",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models (DROID evaluation, Table II)",
     "url": "https://arxiv.org/html/2501.09747",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "RoboArena: Distributed Real-World Evaluation of Generalist Robot Policies (v2)",
     "url": "https://arxiv.org/html/2506.18123v2",
     "type": "paper",
     "publisher": "arXiv (CoRL 2025)",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "RoboArena leaderboard API (all policies)",
     "url": "https://roboarena-api-domain-name.online/api/leaderboard",
     "type": "leaderboard",
     "publisher": "RoboArena",
     "date": "2026-10-11",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (v2; Figures 6-8, 13)",
     "url": "https://arxiv.org/html/2512.16881v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (v3; Table 3, Figure 7)",
     "url": "https://arxiv.org/html/2510.10125",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "REALM: A Real-to-Sim Validated Benchmark for Generalization in Robotic Manipulation (PDF v1, Fig. 6)",
     "url": "https://arxiv.org/pdf/2512.19562v1",
     "type": "paper",
     "publisher": "arXiv (CTU Prague and others)",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation (Table 2)",
     "url": "https://arxiv.org/html/2606.10366v1",
     "type": "paper",
     "publisher": "arXiv (Tsinghua University)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation (v4)",
     "url": "https://arxiv.org/html/2607.01060v4",
     "type": "paper",
     "publisher": "arXiv (KAIST, Config)",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "OpenVLA (Section 3.3, Appendix A; Franka-DROID tests)",
     "url": "https://arxiv.org/html/2406.09246",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "SpatialVLA (DROID removed for final third of pretraining)",
     "url": "https://arxiv.org/html/2501.15830",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "GR00T N1 (DROID among OXE subsets, 23.1M frames)",
     "url": "https://arxiv.org/html/2503.14734",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Isaac-GR00T DROID example (GR00T-N1.7-DROID; open-loop MSE)",
     "url": "https://github.com/NVIDIA/Isaac-GR00T/blob/main/examples/DROID/README.md",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
     "url": "https://arxiv.org/html/2506.09985",
     "type": "paper",
     "publisher": "arXiv (Meta)",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "RobotArena ∞: Scalable Robot Benchmarking via Real-to-Sim Translation",
     "url": "https://arxiv.org/html/2510.23571",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Semantic Scholar API record for arXiv:2403.12945",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2403.12945?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Hugging Face: lerobot/droid_1.0.1 (licence label, downloads)",
     "url": "https://huggingface.co/datasets/lerobot/droid_1.0.1",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Hugging Face: cadene/droid_1.0.1 (licence label, downloads)",
     "url": "https://huggingface.co/datasets/cadene/droid_1.0.1",
     "type": "repo",
     "publisher": "Hugging Face staff account",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Open X-Embodiment dataset spreadsheet (DROID row)",
     "url": "https://docs.google.com/spreadsheets/d/1rPBD77tk60AEIGZrGSODwyyzs5FgCU9Uz3h-3_t2A9g/edit#gid=0",
     "type": "site",
     "publisher": "Open X-Embodiment Collaboration",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "GitHub API: droid-dataset/droid_policy_learning",
     "url": "https://api.github.com/repos/droid-dataset/droid_policy_learning",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "droid issue #80: permission to redistribute derived labels",
     "url": "https://github.com/droid-dataset/droid/issues/80",
     "type": "repo",
     "publisher": "DROID team (issue tracker)",
     "date": "2026-09",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the checked basic entry and research/raw/inventory/core-real.json. Added eight paired validity comparisons, post-release data fixes, raw-data gap, licence gaps, hardware cost, RoboArena use and adoption. Corrected the basic entry: latest update is 2025-09 (annotation repo), not 2025-07; the version-label claim is now verified at the official annotation repo; the paper names 10 and 9 scene types in different places."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "ego-exo4d-proficiency",
   "name": "Ego-Exo4D Proficiency",
   "full_name": "Ego-Exo4D proficiency estimation benchmark",
   "aliases": [
    "Ego-Exo4D proficiency",
    "demonstrator proficiency estimation",
    "demonstration proficiency estimation",
    "EgoExo4D Proficiency Estimation Challenge"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Scores video models on judging human skill; no robot or embodied policy is scored. The scope rule lists Ego-Exo4D as borderline (egocentric human video).",
   "summary": {
    "text": "Video benchmark: estimate a person's skill level, and spot good or weak moments, from first- and third-person video.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "FAIR (Meta) and a university consortium: CMU, CMU Africa, KAUST, U Minnesota, IIIT Hyderabad, Indiana U, UNC Chapel Hill, UT Austin, U Catania, U Tokyo, U Bristol, NUS, Georgia Tech, U Pennsylvania, UIUC, Universidad de los Andes, Simon Fraser U",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Crossref lists 101 authors, first author Kristen Grauman (https://api.crossref.org/works/10.1007/s11263-025-02557-6)."
    },
    "builder_type": {
     "value": "consortium",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Meta FAIR plus 15+ universities."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Partners in North America, Europe, Asia, Africa, South America."
    },
    "first_release": {
     "value": "2023-11 (arXiv v1 2023-11-30; proficiency estimation named in the v1 abstract); dataset release announced 2023-12-15",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Release date from https://docs.ego-exo4d-data.org/changelog/."
    },
    "latest_update": {
     "value": "2025-05: proficiency repo updated the v2 demonstrator test set (commits 2025-05-18 to 2025-05-25); journal version in IJCV published online 2025-11-24",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Commit messages: 'test split video_ids', 'updated v2 test set videos'. IJCV date from https://api.crossref.org/works/10.1007/s11263-025-02557-6."
    },
    "version": {
     "value": "Ego-Exo4D data V2 (2024-03-25); demonstrator_proficiency_test_v2.json",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "CVPR 2024; IJCV vol. 133, issue 12, pp. 8356-8435 (online 2025-11-24)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Crossref metadata (DOI 10.1007/s11263-025-02557-6), article licence CC BY-NC-ND 4.0. CVPR 2024 stated in arXiv comments (https://arxiv.org/abs/2311.18259). Springer and CVF pages blocked automated access."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classification and temporal localisation on recorded video."
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Poor fit; see taxonomy_friction."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Basketball, bike repair, cooking, dance, health (COVID testing), music, rock climbing, soccer."
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Demonstrator proficiency: 1,983 train / 561 val / 768 test takes, 6 scenarios (bike repair and health excluded), 4 classes (novice, early, intermediate, late expert). Demonstration proficiency: 556 train / 177 val / 179 test takes, 8 scenarios.",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "note": "Table 9 of arXiv v4."
      },
      {
       "value": "Parent dataset: 740 participants, 13 cities, 123 scene contexts, 1,286 hours, 5,035 takes (arXiv v4 and V2 data). arXiv v1 said >800 participants, 131 scene contexts, 1,422 hours.",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "note": "Conflict between arXiv v1 (https://arxiv.org/abs/2311.18259v1) and v4; the V2 data release (2024-03-25) lists 1,286.30 hours and 5,035 takes."
      }
     ]
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Docs list mAP at k = 0.25, 0.5, 1.0 s (https://docs.ego-exo4d-data.org/benchmarks/proficiency_estimation/). Challenge numbers are each team's own report (https://arxiv.org/abs/2505.24411, https://arxiv.org/abs/2507.08022); official leaderboard page not found."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "A Proficiency Estimation Challenge ran at the CVPR 2025 EgoVis workshop (participant reports), but no leaderboard page was found. The 2026 challenge page lists only EgoPose Body and Procedure Understanding (CodaBench) and says organisers are moving from EvalAI to CodaBench."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "EGO4D/ego-exo4d-proficiency has no LICENSE file (GitHub API license null). The dataset download tool repo facebookresearch/Ego4d is MIT (https://github.com/facebookresearch/Ego4d)."
    },
    "license_data": {
     "value": "custom: Ego-Exo4D licence agreement",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Signed per individual or institution at ego4d.dev (request page returned HTTP 403 to us). Docs say it covers 'research purposes and commercial use' with restrictions on redistribution; full terms not read."
    },
    "access": {
     "value": "application",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Approval typically ~48 hours; AWS credentials expire after 14 days."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Not applicable in practice: human video only, no robot."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives (full text)",
     "url": "https://arxiv.org/html/2311.18259v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-11"
    },
    "s2": {
     "title": "Ego-Exo4D",
     "url": "https://ego-exo4d-data.org/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives",
     "url": "https://arxiv.org/abs/2311.18259v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-11"
    },
    "s4": {
     "title": "EGO4D/ego-exo4d-proficiency on GitHub (repository)",
     "url": "https://api.github.com/repos/EGO4D/ego-exo4d-proficiency",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "Change Log | Ego-Exo4D Documentation",
     "url": "https://docs.ego-exo4d-data.org/changelog/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "https://api.crossref.org/works/10.1007/s11263-025-02557-6",
     "url": "https://api.crossref.org/works/10.1007/s11263-025-02557-6",
     "type": "index",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "EgoExo4D Challenge 2026 | Ego-Exo4D Documentation",
     "url": "https://docs.ego-exo4d-data.org/challenge/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Getting Started | Ego-Exo4D Documentation",
     "url": "https://docs.ego-exo4d-data.org/getting-started/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2311.18259",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s10": {
     "title": "https://arxiv.org/abs/2311.18259",
     "url": "https://arxiv.org/abs/2311.18259",
     "type": "paper",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "ego4d",
   "name": "Ego4D",
   "aliases": [
    "Ego4D benchmark suite",
    "Ego4D Challenge"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Egocentric human-video dataset and benchmark suite; the scope rule lists egocentric human video benchmarks as borderline. No source found where Ego4D scores a robot or embodied policy; robotics uses it as pre-training data (e.g. R3M). Recommend: list as related dataset, not as a benchmark.",
   "summary": {
    "text": "3,670 hours of first-person human video with a benchmark suite on memory, hand-object interaction, social cues and forecasting.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Site: led by 13 universities in partnership with Facebook AI; 85 authors on arXiv (first author Kristen Grauman)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "builder_type": {
     "value": "consortium",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Data from 74 locations in 9 countries. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2021-10 (arXiv v1 2021-10-13); site gives data release date 2022-02-17",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "CVPR 2022 (arXiv comment)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "v2.1: Goal-Step annotations and grouped videos (210 videos affected). v2.0 (Feb 2023): FHO annotations 243 h vs 120 h, NLQ 27k vs 17.3k queries. 2026 challenges: NLQ, Goal Step, Short-Term object interaction anticipation, at CVPR 2026 EgoVis, using v2.0.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Challenge page: https://ego4d-data.org/docs/challenge/ Checked 2026-10-10."
    },
    "version": {
     "value": "v2.1",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Closest id; tasks are perception and forecasting over human video, not robot reasoning. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Human camera wearers; no robot. Checked 2026-10-10."
    },
    "scale": {
     "value": "3,670 hours; 931 unique camera wearers (arXiv abstract) vs 923 unique participants (site); 74 locations, 9 countries",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Conflict between paper and site participant counts; both recorded. Checked 2026-10-10."
    },
    "scoring": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Official (: challenges historically on EvalAI, moving to CodaBench in 2026)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API licence detection (facebookresearch/Ego4d). Checked 2026-10-10."
    },
    "license_data": {
     "value": "custom: EGO4D License Agreement; must be reviewed and executed before credentials are issued (~48 h); credentials expire after 14 days",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Draft licence PDF: https://ego4d-data.org/pdfs/Ego4D-Licenses-Draft.pdf Checked 2026-10-10."
    },
    "access": {
     "value": "application",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "unclear: draft licence text mentions permitted use for academic research and commercial or noncommercial product development, with restrictions",
     "note": "Read the draft licence PDF; not legal advice. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No evidence that Ego4D benchmark scores predict robot performance. Used as pre-training data for robot representations (R3M)."
    },
    "citations": {
     "value": 2129,
     "display": "2129",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 652,
     "display": "652",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Daily-life activity across 74 locations; no scene breakdown extracted. Checked 2026-10-10."
    },
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Ego4D: Around the World in 3,000 Hours of Egocentric Video",
     "url": "https://arxiv.org/abs/2110.07058",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-10"
    },
    "s2": {
     "title": "Egocentric 4D Perception (EGO4D)",
     "url": "https://ego4d-data.org/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Updates | Ego4D",
     "url": "https://ego4d-data.org/docs/updates/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Ego4D and EgoExo4D Challenge 2026 | Ego4D",
     "url": "https://ego4d-data.org/docs/challenge/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "facebookresearch/Ego4d on GitHub (repository)",
     "url": "https://github.com/facebookresearch/Ego4d",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Start Here | Ego4D",
     "url": "https://ego4d-data.org/docs/start-here/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "https://ego4d-data.org/pdfs/Ego4D-Licenses-Draft.pdf",
     "url": "https://ego4d-data.org/pdfs/Ego4D-Licenses-Draft.pdf",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "R3M: A Universal Visual Representation for Robot Manipulation",
     "url": "https://arxiv.org/abs/2203.12601",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-03"
    },
    "s9": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2110.07058",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "embodiedbench",
   "name": "EmbodiedBench",
   "full_name": "EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents",
   "aliases": [
    "EB-ALFRED",
    "EB-Habitat",
    "EB-Navigation",
    "EB-Manipulation",
    "EmbodiedBench Challenge"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores multimodal LLMs acting closed-loop as embodied agents in simulated 3D environments (household planning, navigation, arm manipulation).",
   "summary": {
    "text": "EmbodiedBench runs multimodal AI models as robot planners in four simulators, from household planning with ready-made skills to navigation and arm control with small movements. It reports the share of its 1,128 tasks completed in each environment.",
    "sources": [
     "s1",
     "s2"
    ],
    "short": "EmbodiedBench is a set of 1,128 simulated tasks in which a multimodal AI model (one that takes in both images and text) plans a robot's actions. It reports the share of tasks the model completes in each of its four environments."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents a fixed set of 1,128 test tasks in four environments, an agent framework and a success metric.",
     "short": "A benchmark of simulated tasks for AI agents"
    },
    "kind_secondary": {
     "value": [
      "challenge"
     ],
     "display": "A CVPR 2026 challenge edition (EB-ALFRED and EB-Navigation) added an organiser-run held-out stage.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "Also a CVPR 2026 challenge"
    },
    "publishers": {
     "value": [
      "University of Illinois Urbana-Champaign",
      "Northwestern University",
      "University of Toronto",
      "Toyota Technological Institute at Chicago"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Senior authors include Heng Ji, Huan Zhang and Tong Zhang (UIUC) and Manling Li (Northwestern).",
     "items": [
      {
       "value": "University of Illinois Urbana-Champaign",
       "display": "Lead institution: 8 of 13 authors, including all three corresponding authors; two more authors interned there.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Northwestern University",
       "display": "Kangrui Wang, Qineng Wang, Manling Li.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "University of Toronto",
       "display": "Mark Zhao (work done during an internship at UIUC).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Toyota Technological Institute at Chicago",
       "display": "Marziyeh Movahedi (work done during an internship at UIUC).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "From author affiliations: four universities and institutes."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Institutions in the United States and Canada; lead institution UIUC."
    },
    "first_release": {
     "value": "2025-02",
     "display": "Repository created 2025-02-12; arXiv v1 2025-02-13. Accepted at ICML 2025 (oral, per the project site).",
     "level": "verified",
     "sources": [
      "s1",
      "s10",
      "s3"
     ],
     "checked": "2026-10-10",
     "short": "February 2025, at ICML 2025"
    },
    "published_at": {
     "value": "ICML 2025",
     "display": "ICML 2025 (arXiv comment 'Accepted to ICML 2025'; project site says oral)",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s23"
     ],
     "checked": "2026-10-10",
     "short": "ICML 2025, as an oral presentation"
    },
    "latest_update": {
     "value": "2026-05-30",
     "display": "2026-05-30: two EB-ALFRED fixes (the find action and ambiguous split instructions). Earlier in 2026: EB-Habitat long-horizon data replaced in place (2026-03-26) and Qwen3-VL support (2026-04-08).",
     "level": "verified",
     "sources": [
      "s5",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "May 2026, with fixes to the environments"
    },
    "version": {
     "value": "paper v3; code 2026-05-30",
     "display": "Paper v3 (2025-06-05) covers 24 models. The code has no tags; results from before the 2026-05-30 EB-ALFRED fixes need commit 5fdee379 to reproduce.",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "EB-ALFRED",
       "display": "300 tasks (6 subsets x 50) from ALFRED in AI2-THOR, using Lota-Bench's simulator code. 8 high-level skill types; 171 to 298 possible actions per task.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "EB-Habitat",
       "display": "300 tasks (6 x 50) from the Language Rearrangement benchmark in Habitat 2.0. 70 high-level skills; 282 instruction templates.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "EB-Navigation",
       "display": "300 tasks (5 subsets x 60) in AI2-THOR, built from 90 tasks (one per scene). Low-level moves, turns and camera tilts.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "EB-Manipulation",
       "display": "228 tasks (48 per subset; visual appearance 36) extending VLMBench in CoppeliaSim. 7-number arm commands, discretised.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "Paper version 3, with four environments"
    },
    "status": {
     "value": "active",
     "display": "Commits through 2026-05-30; the CVPR 2026 challenge ran from April to May 2026. A licence question opened 2026-09-24 (issue #47) has no reply yet.",
     "level": "inferred",
     "sources": [
      "s5",
      "s7",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "Last commit about four months before 2026-10-10.",
     "short": "Active. The last change was in May 2026."
    },
    "capability": {
     "value": [
      "long-horizon",
      "navigation",
      "manipulation",
      "instruction-following",
      "embodied-reasoning"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Order chosen so the Atlas files it under household tasks: two of four environments are household planning. The paper's six capability subsets (base, common sense, complex instruction, spatial awareness, visual appearance, long horizon) are its own categories and only partly match taxonomy values."
    },
    "generalisation": {
     "value": [
      "language",
      "scene-layout",
      "object-instance"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "EB-Habitat's base subset merges Language Rearrangement's 'new scenes', 'novel objects' and 'instruction rephrasing' sets; the common-sense and complex-instruction subsets vary wording. Models are tested without task-specific training, using 10 in-context examples; EmbodiedBench defines no training split.",
     "short": "Instruction wording, rooms and objects vary."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper's Limitations section: evaluation is 'solely in simulated environments, without real-world experiments'."
    },
    "simulator": {
     "value": "AI2-THOR, Habitat 2.0, CoppeliaSim",
     "display": "AI2-THOR (EB-ALFRED via Lota-Bench's code, and EB-Navigation); Habitat 2.0 (EB-Habitat); CoppeliaSim 4.1.0 through VLMBench and PyRep (EB-Manipulation)",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "AI2-THOR, Habitat 2.0 and CoppeliaSim"
    },
    "embodiment": {
     "value": [
      "virtual-agent",
      "mobile-manipulator",
      "mobile-base",
      "single-arm"
     ],
     "level": "inferred",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "EB-ALFRED: AI2-THOR agent with abstract skills (virtual agent, as in the ALFRED entry). EB-Habitat: Fetch robot. EB-Navigation: camera agent that only moves (mobile base). EB-Manipulation: one Franka arm. The mapping is ours."
    },
    "robots": {
     "value": "Fetch (suction gripper); Franka Emika Panda",
     "display": "EB-Habitat: Fetch robot with a suction gripper (config 'FetchSuctionRobot'). EB-Manipulation: 7-DoF Franka Emika Panda arm. EB-ALFRED and EB-Navigation: the AI2-THOR agent, no named robot. All simulated.",
     "level": "verified",
     "sources": [
      "s11",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "Fetch and Franka Panda (simulated)"
    },
    "scene": {
     "value": [
      "home",
      "kitchen",
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "AI2-THOR rooms ('kitchens, living rooms, and bedrooms'), a ReplicaCAD apartment in Habitat, and a table with objects for EB-Manipulation."
    },
    "tasks": {
     "value": 1128,
     "display": "1,128 test tasks: EB-ALFRED 300, EB-Habitat 300, EB-Navigation 300, EB-Manipulation 228",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "EB-ALFRED tasks come from ALFRED's 'valid seen' split, following Lota-Bench.",
     "short": "1,128 tasks in 4 environments"
    },
    "scenes": {
     "value": 90,
     "display": "EB-Navigation draws on 90 AI2-THOR scenes (one base task per scene). Scene counts for the other environments are not stated.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "90 AI2-THOR scenes in EB-Navigation"
    },
    "demonstrations": {
     "value": "no training split; test-task trajectories released",
     "display": "No training split. On 2025-06-03 the authors released trajectory datasets recorded by several models on the benchmark's own tasks, with success flags, and suggest training on the base subset and testing on the others.",
     "level": "verified",
     "sources": [
      "s4",
      "s12"
     ],
     "checked": "2026-10-10",
     "short": "There is no training split. Trajectories recorded on the test tasks have been released."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Task success judged by the simulator (PDDL goal checks in EB-ALFRED and EB-Habitat; reaching within a set distance of the target in EB-Navigation)."
    },
    "metric_detail": {
     "value": "success rate per environment",
     "display": "Share of tasks completed in each environment, shown per capability subset and as an environment average. High-level environments give the model a menu of ready-made skills (for example 'find an apple'); EB-Manipulation's arm commands are executed by a motion planner, with detection boxes and object positions supplied. There is no official overall score across environments.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The abstract's '28.9% on average' for GPT-4o is its EB-Manipulation average, not a four-environment average.",
     "short": "Success rate, averaged within each environment"
    },
    "trials": {
     "value": "one episode per task",
     "display": "Each task is run once at temperature 0, with 10 in-context examples, 500x500 images and step limits of 30 (high-level), 20 (navigation) and 15 (manipulation). Episodes also stop after more than 10 invalid actions or an empty plan.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Chat history in EB-Navigation",
       "display": "Results use chat history by default; turning it off moves scores by about 10 points up or down depending on the model (paper Table 10; maintainer in issue #7).",
       "level": "verified",
       "sources": [
        "s2",
        "s34"
       ]
      },
      {
       "value": "Habitat start states",
       "display": "Initial states vary between runs even with a fixed seed; the maintainers say some randomness is intended (issue #37, open).",
       "level": "verified",
       "sources": [
        "s32"
       ]
      }
     ],
     "short": "One run per task"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Single runs; no error bars in the paper or on the leaderboard. With 300 tasks, a 95% interval is about ±5.7 points at 50% success; a 50-task subset moves in steps of 2 points (our calculation)."
    },
    "evaluator": {
     "value": "both",
     "level": "inferred",
     "sources": [
      "s3",
      "s7",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "The official leaderboard holds the authors' own runs of 24 models. Other papers self-report. The CVPR 2026 challenge's final stage was run by the organisers on held-out tasks."
    },
    "leaderboard": {
     "value": "official",
     "display": "Project-site leaderboard: 24 models per environment, all from the paper (data files last modified 2026-06-02). Separate challenge boards for the open split and the held-out stage.",
     "level": "verified",
     "sources": [
      "s3",
      "s6",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The leaderboard values equal paper Tables 2 and 3, so they were computed before the 2026 environment fixes (our comparison).",
     "short": "Official. It lists 24 models, all with results from the paper."
    },
    "human_baseline": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No human results in the paper (full text searched), README or project site."
    },
    "top_score": {
     "value": "67.7 / 68.0 / 57.7 / 28.9",
     "display": "Best on the official leaderboard: EB-ALFRED 67.7 (Claude-3.7-Sonnet), EB-Habitat 68.0 (Claude-3.5-Sonnet), EB-Navigation 57.7 (GPT-4o), EB-Manipulation 28.9 (GPT-4o). Later papers report higher numbers under other protocols (items). There is no single headline score, so there is no chart.",
     "level": "verified",
     "sources": [
      "s6",
      "s24",
      "s25",
      "s26",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ERA-3B (EPL+RL), 2025-10",
       "display": "EB-ALFRED 65.2 and EB-Manipulation 48.3, five-subset averages. Trained on trajectories from three test subsets it calls 'seen' (base, complex, visual). Same team as EmbodiedBench.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      },
      {
       "value": "GPT-4o / Gemini-1.5-pro + BrainMem, 2026-03",
       "display": "EB-ALFRED 75.0 and 75.3 (six subsets) with a memory add-on, against 56.3 and 62.3 without it.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      },
      {
       "value": "Mimir memory system, 2026-08",
       "display": "With Qwen2.5-VL-72B: EB-Habitat 70.0 (four-subset average). Qwen3.5-27B alone: EB-ALFRED 70.5 (four-subset average).",
       "level": "verified",
       "sources": [
        "s16"
       ]
      },
      {
       "value": "CVPR 2026 challenge winner (NJU-LAMDA-SZ)",
       "display": "Held-out stage: EB-ALFRED 67.3, EB-Navigation 37.0, average 52.2, with an open model under 10B parameters. On the public tasks the same team scored 82.50.",
       "level": "verified",
       "sources": [
        "s7"
       ]
      }
     ],
     "short": "The best scores are 67.7 on EB-ALFRED, 68.0 on EB-Habitat, 57.7 on EB-Navigation and 28.9 on EB-Manipulation."
    },
    "license_code": {
     "value": "none stated",
     "display": "No LICENSE file and no licence statement in the README. A user asked for clarification on 2026-09-24 (issue #47, no reply yet).",
     "level": "verified",
     "sources": [
      "s10",
      "s4",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub licence endpoint returns 404. Without a licence, default copyright applies (our reading; not legal advice).",
     "short": "No licence is stated."
    },
    "license_data": {
     "value": [
      "Apache-2.0",
      "none stated"
     ],
     "display": "EB-ALFRED dataset card: Apache-2.0 (derived from ALFRED). EB-Manipulation card and the four trajectory datasets: no licence.",
     "level": "verified",
     "sources": [
      "s17",
      "s35",
      "s12"
     ],
     "checked": "2026-10-10",
     "short": "Apache-2.0 for EB-ALFRED. The other datasets state no licence."
    },
    "license_assets": {
     "value": "mixed",
     "display": "The environments build on third-party simulators and data with their own terms.",
     "level": "inferred",
     "sources": [
      "s18",
      "s19",
      "s20",
      "s21",
      "s22",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "ReplicaCAD and YCB terms were not checked. Not legal advice.",
     "items": [
      {
       "value": "AI2-THOR: Apache-2.0",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "ALFRED: MIT",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "habitat-lab: MIT",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Lota-Bench code (LLMTaskPlanning): no licence",
       "display": "The repository whose simulator code EB-ALFRED builds on has no licence file.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "CoppeliaSim: commercial licence needed",
       "display": "The README installs CoppeliaSim Pro 4.1.0. Coppelia Robotics lists commercial use under paid licences; its free Edu edition is limited to schools and universities.",
       "level": "verified",
       "sources": [
        "s22",
        "s4"
       ]
      }
     ],
     "short": "Mixed. CoppeliaSim needs a paid licence for commercial use."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s4",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub; datasets on Hugging Face without a gate. EB-Manipulation needs CoppeliaSim; EB-Habitat needs Habitat's YCB and ReplicaCAD downloads.",
     "short": "Open. The code is on GitHub and the data is on Hugging Face."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s10",
      "s35",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "The code has no licence, so no rights are granted by default (our reading). EB-ALFRED data is Apache-2.0; other data has no licence. EB-Manipulation needs CoppeliaSim, which requires a paid licence for commercial use. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "Simulation only; no study compares EmbodiedBench scores with real-robot results.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s7",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "The paper's Limitations section says evaluation is solely in simulation. Vlaser (ICLR 2026) found that gains on embodied-reasoning benchmarks, EB-ALFRED and EB-Habitat among them, did not carry over to closed-loop robot control in SimplerEnv, which is also simulation.",
     "short": "We found no study that compares it with real robots."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "citations": {
     "value": 256,
     "display": "256 (Semantic Scholar; 21 influential)",
     "level": "verified",
     "sources": [
      "s23"
     ],
     "checked": "2026-10-10",
     "short": "256"
    },
    "github_stars": {
     "value": 347,
     "display": "347 stars, 42 forks (EmbodiedBench/EmbodiedBench)",
     "level": "verified",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10",
     "short": "347"
    },
    "dataset_downloads": {
     "value": 718,
     "display": "EB-Manipulation 718 and EB-ALFRED 702 (Hub 'downloads' field); trajectory datasets 108 to 169 each",
     "level": "verified",
     "sources": [
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hugging Face API on 2026-10-10. We did not check the time window behind the 'downloads' field.",
     "short": "About 700 for each of the two main datasets"
    },
    "used_by": {
     "value": "Used mainly in academic agent papers; we found no company model report with EmbodiedBench results.",
     "level": "inferred",
     "sources": [
      "s27",
      "s14",
      "s15",
      "s16",
      "s28",
      "s7"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Vlaser",
       "display": "2025-10, ICLR 2026. Reports EB-ALFRED and EB-Habitat (Vlaser-8B: 50.0 and 40.0).",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "ERA",
       "display": "2025-10, same team. Trains a 3B model on benchmark trajectories plus RL.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      },
      {
       "value": "BrainMem",
       "display": "2026-03. Memory add-on for planners.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      },
      {
       "value": "Mimir",
       "display": "2026-08. Memory system tested on 12 backbones.",
       "level": "verified",
       "sources": [
        "s16"
       ]
      },
      {
       "value": "Embodied Arena",
       "display": "2025-09. Evaluation platform that includes EB-ALFRED, EB-Habitat and EB-Navigation.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "CVPR 2026 challenge",
       "display": "Five teams reached the final; four finished the held-out stage.",
       "level": "verified",
       "sources": [
        "s7"
       ]
      }
     ],
     "short": "Academic agent papers. We found no company reports that use it."
    },
    "industry_use": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No company model report with EmbodiedBench results found. Checked HY-Embodied-0.5, Seed1.8, Qwen3-VL, InternVL3.5, GLM-4.5V, Xiaomi-Robotics-0 and Wall-OSS-0.5 (none mention it) and searched for Pelican-VL, RoboBrain 2.5, MiMo-Embodied, Embodied-R1.5 and EmbodiedBrain (no EmbodiedBench results in retrieved text). A challenge team is named 'Ideal-Embody(Ideal)'; its affiliation is not stated."
    },
    "derived_benchmarks": {
     "value": [
      "EmbodiedBench Challenge (CVPR 2026)"
     ],
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "EmbodiedBench Challenge (CVPR 2026)",
       "display": "EB-ALFRED and EB-Navigation; open models under 10B parameters, no commercial APIs; Stage 1 on public tasks (2026-04-15 to 05-25), Stage 2 on held-out tasks run by the organisers. Held-out tasks to be released after the challenge.",
       "level": "verified",
       "sources": [
        "s7"
       ]
      }
     ],
     "short": "A challenge edition at CVPR 2026"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "Fixes in 2026 changed the test",
     "text": "Three fixes changed results. On 2026-03-26, EB-Habitat long-horizon instructions that named the wrong starting places for objects were rewritten and the data replaced in place (issues #28 and #39). On 2026-05-30, EB-ALFRED's 'find' action was fixed: it could teleport the agent out of the room or report success when the target was not visible (issue #43, reported 2026-05-06; the fix was held back until the challenge's first stage ended). Split instructions with ambiguous 'bottle' wording were corrected the same day. Earlier results need commit 5fdee379 to reproduce. The official leaderboard still shows the paper's pre-fix numbers.",
     "level": "verified",
     "sources": [
      "s4",
      "s13",
      "s33",
      "s6"
     ],
     "status": "open",
     "short": "Fixes in 2026 changed EB-ALFRED and EB-Habitat. Scores from before the fixes used the old environments."
    },
    {
     "id": "i2",
     "type": "shortcut",
     "title": "The household tasks barely need the images",
     "text": "In the paper's own test, GPT-4o without images scored 58.0 on EB-ALFRED against 56.3 with images, and 56.0 against 59.0 on EB-Habitat; GPT-4o-mini did better without images on both (31.3 vs 24.0 and 36.7 vs 32.7). Removing images hurt the low-level environments: GPT-4o fell from 57.7 to 17.4 on EB-Navigation and from 28.9 to 16.2 on EB-Manipulation. The authors conclude the high-level tasks may rely more on text than on vision. In EB-ALFRED, the 'find' skill moves the agent to the named object without the model needing to see it.",
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "status": "open",
     "short": "In the paper's own test, GPT-4o scored 58.0 without images and 56.3 with images on EB-ALFRED, and 56.0 without and 59.0 with on EB-Habitat. Removing images hurt the navigation and manipulation tasks much more."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "Papers average different subsets, and some published numbers do not reproduce",
     "text": "The original paper averages six subsets (five for navigation and manipulation). ERA reports five-subset averages for EB-ALFRED, which moves Claude-3.5-Sonnet from 64.0 to 66.4; Mimir averages four subsets; RoboMemory-style comparisons average only base and long-horizon. The authors' Qwen2.5-VL scores came from Alibaba's API, where some requests were blocked; they advise relying on self-hosted runs (issues #29 and #33; one user got 0.52 on EB-ALFRED long-horizon for Qwen2.5-VL-72B against 0.34 published). The paper's text names Claude-3.5-Sonnet best on EB-ALFRED at 64.0 while its Table 2 shows Claude-3.7-Sonnet at 67.7, and the project page still says 13 models were evaluated where the paper says 24.",
     "level": "verified",
     "sources": [
      "s2",
      "s14",
      "s16",
      "s15",
      "s30",
      "s31",
      "s3"
     ],
     "status": "open",
     "short": "Papers average over 2, 4, 5 or 6 subsets, so their numbers are rarely comparable."
    },
    {
     "id": "i4",
     "type": "contamination",
     "title": "Test tasks and solved trajectories are public",
     "text": "There is no hidden test set. The authors released trajectory datasets recorded on the benchmark's own tasks, with success flags, and suggest training on the base subset and testing on others. ERA, by the same team, trains on trajectories from three test subsets it calls 'seen' and includes them in its averages. EB-ALFRED tasks come from ALFRED's 'valid seen' split, whose rooms also appear in ALFRED's training data. In the CVPR 2026 challenge, the winning team scored 82.50 on the public tasks and 52.2 on organiser-run held-out tasks (EB-Navigation 81.67 to 37.0), though the organisers say the held-out navigation tasks were harder, and they raised the step limit to 30.",
     "level": "verified",
     "sources": [
      "s12",
      "s4",
      "s14",
      "s2",
      "s7"
     ],
     "status": "open",
     "short": "Solved trajectories for the test tasks are public. The winner of the CVPR 2026 challenge scored 82.5 on the public tasks and 52.2 on held-out tasks."
    },
    {
     "id": "i5",
     "type": "protocol-variance",
     "title": "The agent framework changes scores a lot",
     "text": "A score measures a model inside an agent framework. Adding a memory module raised GPT-4o on EB-ALFRED from 56.3 to 75.0 (BrainMem). Mimir raised InternVL3-8B from 17.5 to 60.0 (four-subset average). In the paper, removing environment feedback cost GPT-4o about 10 points on the EB-ALFRED base subset, and using no in-context examples dropped success to about 40%. Chat-history settings move EB-Navigation results by about 10 points in either direction.",
     "level": "verified",
     "sources": [
      "s15",
      "s16",
      "s2",
      "s34"
     ],
     "status": "open",
     "short": "A score depends on the agent framework (the prompts, examples and add-ons around the model). A memory add-on raised GPT-4o from 56.3 to 75.0 on EB-ALFRED."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "The code has no licence",
     "text": "The repository has no LICENSE file and the README states no licence (GitHub API, 2026-10-10). On 2026-09-24 a group preparing an ICLR submission asked whether running the code and reporting scores is permitted; there is no reply yet (issue #47). Only the EB-ALFRED data card states a licence (Apache-2.0). EB-Manipulation also requires CoppeliaSim, which needs a paid licence for commercial use.",
     "level": "verified",
     "sources": [
      "s10",
      "s29",
      "s17",
      "s22"
     ],
     "status": "open",
     "short": "The code has no licence. A user asked for clarification and has had no reply."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "An EmbodiedBench score measures a whole agent. The agent is the model together with the authors' prompt, ten example plans and the simulator's ready-made skills. Changing only the agent framework moved GPT-4o from 56.3 to 75.0 on EB-ALFRED. Compare models only under the same framework and code version.",
     "basis": [
      "issues.i5",
      "issues.i1",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A score reflects both the model and the framework around it. Compare models only under the same framework."
    },
    {
     "id": "r2",
     "text": "EB-ALFRED and EB-Habitat scores say more about planning from text than about seeing. GPT-4o did as well on them without images. Use EB-Navigation and EB-Manipulation to judge how much a model uses vision.",
     "basis": [
      "issues.i2"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Use the navigation and arm tasks to judge how much a model uses vision."
    },
    {
     "id": "r3",
     "text": "Treat high scores from models trained on EmbodiedBench trajectories with caution. The test tasks are public, and the challenge's held-out stage showed large drops, although its navigation tasks were also harder.",
     "basis": [
      "issues.i4",
      "facts.demonstrations"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Training on public test trajectories can inflate scores."
    },
    {
     "id": "r4",
     "text": "None of these results show real-robot ability. Actions are high-level skills or discretised arm poses carried out by a motion planner, and the benchmark has never been compared with real robots.",
     "basis": [
      "facts.sim_to_real",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The benchmark runs only in simulation. It gives no evidence about real robots."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a model will do on a real robot.",
     "sub": "EmbodiedBench runs only in simulation. No study has compared its scores with real-robot results.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a model can control a robot's fine movements.",
     "sub": "The model picks from ready-made skills or gives arm poses that a motion planner carries out.",
     "basis": [
      "facts.metric_detail",
      "facts.robots"
     ]
    },
    {
     "id": "l3",
     "text": "How good the model is on its own.",
     "sub": "Scores change with the prompts, the examples and any memory add-ons used around the model.",
     "basis": [
      "issues.i5"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Paper (Limitations and full text), README, project site, challenge page; Vlaser (2510.11027); web searches on 2026-10-10 for real-robot evaluations of EmbodiedBench tasks or correlations with real-robot results. None found.",
     "date": "2026-10-10"
    },
    {
     "for": "human_baseline",
     "where": "Paper full text, README, project site. None.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code",
     "where": "GitHub licence API (404), repository root listing (no LICENSE), README, issue #47.",
     "date": "2026-10-10"
    },
    {
     "for": "industry_use",
     "where": "Full texts of HY-Embodied-0.5, Seed1.8, Qwen3-VL, InternVL3.5, GLM-4.5V, Xiaomi-Robotics-0, Wall-OSS-0.5 (no EmbodiedBench); web search for Pelican-VL, RoboBrain 2.5, MiMo-Embodied, Embodied-R1.5, EmbodiedBrain (no EmbodiedBench results in retrieved text); Athena-Brain report (no mention).",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "Official leaderboard JSON files; ERA, BrainMem, Mimir, Vlaser; challenge page. Higher numbers exist only under other protocols (items).",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (ReplicaCAD, YCB)",
     "where": "Not checked; VLMBench repository not found at the path tried.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "EmbodiedBench (arXiv abstract page, v3)",
     "url": "https://arxiv.org/abs/2502.09560",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "EmbodiedBench full text v3 (Sections 3 to 6, Tables 2, 3 and 10, Appendix C, Limitations)",
     "url": "https://arxiv.org/html/2502.09560v3",
     "type": "paper",
     "publisher": "arXiv (ICML 2025)",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "EmbodiedBench project site",
     "url": "https://embodiedbench.github.io/",
     "type": "site",
     "publisher": "EmbodiedBench team",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "EmbodiedBench README (news, update notes, installation)",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/blob/master/README.md",
     "type": "repo",
     "publisher": "EmbodiedBench",
     "date": "2026-05-30",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "EmbodiedBench commit history",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/commits/master",
     "type": "repo",
     "publisher": "EmbodiedBench",
     "date": "2026-05-30",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "Official leaderboard data: EB-ALFRED",
     "url": "https://embodiedbench.github.io/website/data/eb_alfred_total_benchmark.json",
     "type": "leaderboard",
     "publisher": "EmbodiedBench team",
     "date": "2026-06-02",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "EmbodiedBench Challenge @ CVPR 2026 (rules, open and held-out leaderboards)",
     "url": "https://embodiedbench.github.io/challenge.html",
     "type": "site",
     "publisher": "Foundation Models Meet Embodied Agents Workshop",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "GitHub API: EmbodiedBench/EmbodiedBench (stars, forks, created, licence)",
     "url": "https://api.github.com/repos/EmbodiedBench/EmbodiedBench",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "EB-Habitat task config (articulated_agent_type: FetchSuctionRobot)",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/blob/master/embodiedbench/envs/eb_habitat/config/task/task_obs/visual.yaml",
     "type": "repo",
     "publisher": "EmbodiedBench",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "EB-Alfred trajectory dataset card",
     "url": "https://huggingface.co/datasets/EmbodiedBench/EB-Alfred_trajectory_dataset",
     "type": "repo",
     "publisher": "EmbodiedBench (Hugging Face)",
     "date": "2025-06-04",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "EmbodiedBench issue #43: find can report success when the target is not visible",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/43",
     "type": "repo",
     "publisher": "EmbodiedBench (user report, maintainer replies)",
     "date": "2026-05-06",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "ERA: Transforming VLMs into Embodied Agents via Embodied Prior Learning and Online Reinforcement Learning (Table 3)",
     "url": "https://arxiv.org/abs/2510.12693",
     "type": "paper",
     "publisher": "arXiv (EmbodiedBench team)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "BrainMem: Brain-Inspired Evolving Memory for Embodied Agent Task Planning (Tables 1 and 3)",
     "url": "https://arxiv.org/abs/2604.16331",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Mimir: A Neuro-Symbolic Memory System with Dynamic Grounding for Embodied Agents (Tables 1 and 2)",
     "url": "https://arxiv.org/abs/2608.04933",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Hugging Face API: EmbodiedBench datasets (licences, downloads)",
     "url": "https://huggingface.co/api/datasets?author=EmbodiedBench&full=true",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "AI2-THOR repository (Apache-2.0)",
     "url": "https://github.com/allenai/ai2thor",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "ALFRED repository (MIT)",
     "url": "https://github.com/askforalfred/alfred",
     "type": "repo",
     "publisher": "ALFRED authors",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "habitat-lab repository (MIT)",
     "url": "https://github.com/facebookresearch/habitat-lab",
     "type": "repo",
     "publisher": "Meta AI",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "LLMTaskPlanning (Lota-Bench) repository (no licence)",
     "url": "https://github.com/lbaa2022/LLMTaskPlanning",
     "type": "repo",
     "publisher": "Lota-Bench authors",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Coppelia Robotics site (editions and commercial licensing)",
     "url": "https://www.coppeliarobotics.com/",
     "type": "site",
     "publisher": "Coppelia Robotics",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Semantic Scholar API record for arXiv:2502.09560",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2502.09560?fields=title,citationCount,influentialCitationCount,externalIds,publicationDate,venue",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Official leaderboard data: EB-Habitat",
     "url": "https://embodiedbench.github.io/website/data/eb_habitat_total_benchmark.json",
     "type": "leaderboard",
     "publisher": "EmbodiedBench team",
     "date": "2026-06-02",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Official leaderboard data: EB-Navigation",
     "url": "https://embodiedbench.github.io/website/data/eb_navigation_total_benchmark.json",
     "type": "leaderboard",
     "publisher": "EmbodiedBench team",
     "date": "2026-06-02",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Official leaderboard data: EB-Manipulation",
     "url": "https://embodiedbench.github.io/website/data/eb_manipulation_total_benchmark.json",
     "type": "leaderboard",
     "publisher": "EmbodiedBench team",
     "date": "2026-06-02",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Vlaser (Table 1; Section 3.2)",
     "url": "https://arxiv.org/abs/2510.11027",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Embodied Arena: A Comprehensive, Unified, and Evolving Evaluation Platform for Embodied AI (Table 1)",
     "url": "https://arxiv.org/abs/2509.15273",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "EmbodiedBench issue #47: License clarification",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/47",
     "type": "repo",
     "publisher": "EmbodiedBench (user request)",
     "date": "2026-09-24",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "EmbodiedBench issue #29: Unable to reproduce scores for Qwen2.5-VL-7B-Instruct",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/29",
     "type": "repo",
     "publisher": "EmbodiedBench (maintainer reply)",
     "date": "2025-10-27",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "EmbodiedBench issue #33: higher task success for Qwen2.5-VL-72B on EB-ALFRED long-horizon",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/33",
     "type": "repo",
     "publisher": "EmbodiedBench (maintainer reply)",
     "date": "2025-11-20",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "EmbodiedBench issue #37: non-deterministic initial states in Habitat",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/37",
     "type": "repo",
     "publisher": "EmbodiedBench (maintainer reply)",
     "date": "2026-02-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "EmbodiedBench issue #39: instruction and scene mismatch in EB-Habitat long-horizon tasks",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/39",
     "type": "repo",
     "publisher": "EmbodiedBench (maintainer reply)",
     "date": "2026-03-25",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "EmbodiedBench issue #7: Test result (chat history setting)",
     "url": "https://github.com/EmbodiedBench/EmbodiedBench/issues/7",
     "type": "repo",
     "publisher": "EmbodiedBench (maintainer replies)",
     "date": "2025-03-15",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "EB-ALFRED dataset card (Apache-2.0)",
     "url": "https://huggingface.co/datasets/EmbodiedBench/EB-ALFRED",
     "type": "repo",
     "publisher": "EmbodiedBench (Hugging Face)",
     "date": "2025-02-22",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Basic entry created (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Full entry written from primary sources, starting from the basic entry and the core-sim-b inventory record; every fact re-checked. Added environment details, robot models (Fetch from the EB-Habitat config; Franka Panda), the vision ablation, 2026 fixes and GitHub issues, the CVPR 2026 challenge open vs held-out results, later protocol variants (ERA, BrainMem, Mimir), trajectory datasets and licence findings. Confirmed prior leads: 1,128 tasks, 24 models, ICML 2025, no LICENSE file, Apache-2.0 for EB-ALFRED only, 256 citations, 347 stars, the 13-vs-24 model count conflict on the project page. Basic entry's evaluator value (null) replaced by 'both'."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "embodiedgovbench",
   "name": "EmbodiedGovBench",
   "aliases": [
    "EmbodiedGovBench",
    "GovScore",
    "AEROS P7"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Embodied safety/governance benchmark that scores embodied agent systems in a 3D household simulator.",
   "summary": {
    "text": "Scores embodied agent systems on governance: permission limits, recovery, upgrades, human override and audit trails, in AI2-THOR scenarios.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Harbin Institute of Technology (3 authors), Heriot-Watt University Malaysia campus, Soochow University.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Repo README citation adds a sixth author (Zeyd Boukhers) not on arXiv v1."
    },
    "first_release": {
     "value": "2026-04 (arXiv v1 2026-04-13).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-05-05: single-commit code release 'EmbodiedGovBench v0.1 (paper-submission artifact bundle)', tag v0.1.0-jair-submission.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Tag name suggests a JAIR submission; acceptance not found."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "safety"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "125 scenario instances (25 per protocol, protocols A, B, C, E, F), 4 systems, seed 42; run took ~6 h on one A100.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Protocol D (portability) and the fleet track were not run."
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked in paper text and repo README."
    },
    "scoring": {
     "value": [
      "composite",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Authors: weights 'not empirically calibrated in the current design'."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (None; results in paper and repo README.)"
    },
    "license_code": {
     "value": "Apache-2.0 (LICENSE file).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Scenario configs and traces are in the same repo under Apache-2.0.",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (None found; authors call external validity undemonstrated.)",
     "note": "Authors state external validity (prediction of real deployment governance) is undemonstrated."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "EmbodiedGovBench: A Benchmark for Governance, Recovery, and Upgrade Safety in Embodied Agent Systems",
     "url": "https://arxiv.org/abs/2604.11174",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-04"
    },
    "s2": {
     "title": "EmbodiedGovBench: A Benchmark for Governance, Recovery, and Upgrade Safety in Embodied Agent Systems (full text)",
     "url": "https://arxiv.org/html/2604.11174v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-04"
    },
    "s3": {
     "title": "AEROS Project — Embodied Agent Runtime Research Program",
     "url": "https://s20sc.github.io/aeros-project/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "s20sc/embodied-gov-bench on GitHub (repository)",
     "url": "https://github.com/s20sc/embodied-gov-bench",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "s20sc/embodied-gov-bench on GitHub (file run_pilot.py)",
     "url": "https://github.com/s20sc/embodied-gov-bench/blob/main/run_pilot.py",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2604.11174",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "erqa",
   "name": "ERQA",
   "full_name": "Embodied Reasoning Question Answering (ERQA), from the Gemini Robotics report",
   "aliases": [
    "Embodied Reasoning QA",
    "Embodied Reasoning Question Answer",
    "ERQA benchmark"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "An embodied-reasoning question set built for robots and released openly. The scope rule includes these.",
   "summary": {
    "text": "ERQA is a set of 400 multiple-choice questions about images from robots and first-person video, released by Google DeepMind in March 2025. It scores vision-language models, AI models that read images and text, on reasoning about space, actions and tasks. No robot moves.",
    "sources": [
     "s2",
     "s4"
    ],
    "short": "ERQA is a set of 400 multiple-choice questions about images from robots and first-person video. It scores how well vision-language models (AI models that read images and text) answer them, and no robot moves."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The report calls ERQA a benchmark. The repository ships one fixed test file with answers and an evaluation script.",
     "short": "A benchmark with one fixed test set"
    },
    "publishers": {
     "value": [
      "Google DeepMind"
     ],
     "display": "Google DeepMind (Gemini Robotics Team)",
     "level": "verified",
     "sources": [
      "s4",
      "s1",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "README: released as part of Google DeepMind's Gemini Robotics release. The report is authored by the Gemini Robotics Team. All repository commits are by Ted Xiao, a report author.",
     "short": "Google DeepMind"
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Built and released by the robotics team of an AI lab."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s1",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The report and the repository state no location. Coded multi to match the basic entry; an earlier Atlas coder chose europe. The deepmind.google careers and about pages opened on 2026-10-10 list no offices in their text."
    },
    "first_release": {
     "value": "2025-03",
     "display": "Repository created 2025-03-07; test file added 2025-03-08; announced with Gemini Robotics on 2025-03-12; report on arXiv 2025-03-25.",
     "level": "verified",
     "sources": [
      "s7",
      "s8",
      "s11",
      "s1",
      "s41"
     ],
     "checked": "2026-10-10",
     "note": "The report is an arXiv technical report, not a peer-reviewed paper.",
     "short": "March 2025, with Gemini Robotics"
    },
    "latest_update": {
     "value": "2025-03-12",
     "display": "Last commit 2025-03-12 (adds links to the blog post and report). The test file has not changed since 2025-03-08.",
     "level": "verified",
     "sources": [
      "s8",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API pushed_at: 2025-03-12T19:53:12Z. No tags or releases.",
     "short": "March 2025. Nothing has changed since then."
    },
    "version": {
     "value": "single release",
     "display": "One unversioned release: data/erqa.tfrecord (91,402,921 bytes) holding the 400 questions, an example loader and an evaluation script.",
     "level": "verified",
     "sources": [
      "s4",
      "s42"
     ],
     "checked": "2026-10-10",
     "note": "File size from the GitHub git-tree API. We did not download the file; the count of 400 comes from the README and the report.",
     "short": "One release with 400 questions"
    },
    "status": {
     "value": "dormant",
     "display": "No repository change since 2025-03-12. The five open issues (2025-06 to 2026-07) have no maintainer reply. Use keeps growing: Google reported ERQA again for Gemini Robotics ER 2 in July 2026.",
     "level": "inferred",
     "sources": [
      "s8",
      "s9",
      "s13",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "Last commit is 19 months before 2026-10-10. The only reply on any open issue (#2) is from another user.",
     "short": "No changes to the repository since March 2025. It is still in active use."
    },
    "published_at": {
     "value": "arXiv technical report",
     "display": "Introduced in the Gemini Robotics technical report (arXiv 2503.20020). No peer-reviewed venue.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "short": "arXiv technical report"
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The report names the categories: spatial reasoning, trajectory reasoning, action reasoning, state estimation, pointing, multi-view reasoning and task reasoning."
    },
    "generalisation": {
     "value": [
      "none-stated"
     ],
     "level": "inferred",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "ERQA is one fixed test set with no training split. The report describes no controlled change between training and test conditions.",
     "short": "None stated. ERQA is a test set only."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Models answer questions about still images. Nothing runs in a simulator or on a robot."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Images show robot arms and people's hands. The model being tested controls nothing."
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "display": "Robot lab and household scenes, and first-person frames of people at work",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Images are the authors' own or come from Open X-Embodiment (OXE), UMI Data, MECCANO, HoloAssist and EGTEA Gaze+ (report Section 2.1). The report gives no breakdown by source.",
     "short": "A mix of robot scenes and first-person scenes"
    },
    "tasks": {
     "value": 400,
     "display": "400 multiple-choice questions in seven named categories plus Other: spatial reasoning 84, action reasoning 72, trajectory reasoning 66, state estimation 55, task reasoning 38, multi-view reasoning 37, pointing 34, other 14.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The category counts are printed in the report's Figure 4. We read them from the figure's source file (erqa_categories.svg) and matched each number to its label by position. They sum to 400.",
     "short": "400 questions in 7 categories plus Other"
    },
    "scale": {
     "value": "28% of questions use more than one image",
     "display": "Each question mixes text and one or more images. 28% of questions have more than one image, and the report says these tend to be harder. Answers are a single letter, A to D (README).",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "28% of questions use several images"
    },
    "demonstrations": {
     "value": "none",
     "display": "No training data. The repository holds only the 400-question test file.",
     "level": "verified",
     "sources": [
      "s4",
      "s42"
     ],
     "checked": "2026-10-10",
     "short": "None. ERQA is a test set only."
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Multiple-choice accuracy (report Table 1)."
    },
    "metric_detail": {
     "value": "percent correct",
     "display": "Accuracy over the 400 questions. The released script counts an answer as correct only when the model's whole reply, with periods and spaces removed and case ignored, equals the answer letter. In the Gemini Robotics 1.5 report, Gemini 2.5 Flash graded the answers instead. The first report also gives results with a chain-of-thought instruction added to each question.",
     "level": "verified",
     "sources": [
      "s6",
      "s12",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Under the released script, a model that explains its answer is marked wrong unless a grader or an answer-extraction step is added.",
     "short": "Percent of 400 questions answered correctly"
    },
    "trials": {
     "value": "one pass over 400 questions",
     "display": "Each question is asked once. The script calls the model API at temperature 0. Gemini Robotics 1.5 used default thinking budgets and no tools.",
     "level": "verified",
     "sources": [
      "s6",
      "s12"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Script default: 1 question",
       "display": "The script's --num_examples default is 1, while the README says the full benchmark is 400 examples. A run with default settings scores one question.",
       "level": "verified",
       "sources": [
        "s6",
        "s4"
       ]
      },
      {
       "value": "GR 1.5 query window",
       "display": "Gemini 2.5 and GPT-5 models were queried between 2025-09-01 and 2025-09-20.",
       "level": "verified",
       "sources": [
        "s12"
       ]
      }
     ],
     "short": "One pass over the 400 questions"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s12",
      "s13",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "All reports we opened give single accuracies without intervals. Gemini Robotics 1.5 averages 3 runs only in its thinking-budget plots (Fig. 16). With 400 questions, a 95% binomial interval is about ±4.9 points at 50% accuracy and ±4.2 points at 75% (our calculation: 1.96 x sqrt(p(1-p)/400))."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s12",
      "s13",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Each model team runs ERQA itself. In its comparisons Google also runs other companies' models through their public APIs (for example GPT-5, Claude Opus 4.5, Opus 5 and GPT 5.6 Sol)."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s2",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard in the repository or in Google's reports; results live in model reports. Embodied Arena (arXiv 2509.15273) describes community leaderboards that include ERQA, but its site is a script-only page we could not read on 2026-10-10. The EASI community board has no ERQA column.",
     "short": "Results appear only in papers."
    },
    "human_baseline": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No human score in the Gemini Robotics report (Section 2.1, Tables 1 and 2), the Gemini Robotics 1.5 report (Appendix C), the README, or the ER 2 post and model card."
    },
    "top_score": {
     "value": 78.5,
     "display": "78.5% by Gemini Robotics ER 2 (Google chart, 2026-07-30). Highest result we found. The rows below come from different reports with different graders and settings, so they are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Each number is verified at its source; 'highest' is our judgement. Google built ERQA and reports most of the top scores; the ER 2 chart states no grader or settings. There is no human baseline to compare against.",
     "items": [
      {
       "value": 48.3,
       "display": "Gemini 2.0 Pro Experimental, results obtained Feb 2025, published Mar 2025. Best at release. With a chain-of-thought prompt it scored 54.8.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "data": {
        "model": "Gemini 2.0 Pro Experimental",
        "date": "2025-03",
        "avg": 48.3,
        "rl": false
       }
      },
      {
       "value": 46.8,
       "display": "InternVL3.5-241B-A28B (open weights, Shanghai AI Laboratory), 2025-08.",
       "level": "verified",
       "sources": [
        "s19"
       ],
       "data": {
        "model": "InternVL3.5-241B-A28B",
        "date": "2025-08",
        "avg": 46.8,
        "rl": false
       }
      },
      {
       "value": 59,
       "display": "GPT-5, 2025-10, run by Google with Gemini 2.5 Flash as grader. InternVL3.5's own run of GPT-5 (VLMEvalKit) gave 65.7.",
       "level": "verified",
       "sources": [
        "s12",
        "s19"
       ],
       "data": {
        "model": "GPT-5 (run by Google)",
        "date": "2025-10",
        "avg": 59,
        "rl": false
       }
      },
      {
       "value": 54.8,
       "display": "Gemini Robotics-ER 1.5 with thinking, 2025-10 (47.0 without thinking). Same table: Gemini 2.5 Pro 56.0.",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "Gemini Robotics-ER 1.5",
        "date": "2025-10",
        "avg": 54.8,
        "rl": false
       }
      },
      {
       "value": 52.5,
       "display": "Qwen3-VL-235B-A22B (Qwen Team), 2025-11.",
       "level": "verified",
       "sources": [
        "s17"
       ],
       "data": {
        "model": "Qwen3-VL-235B-A22B",
        "date": "2025-11",
        "avg": 52.5,
        "rl": false
       }
      },
      {
       "value": 70.5,
       "display": "Gemini 3 Pro, 2025-12, Google's vision benchmark table. HY-Embodied-0.5's own API run in March 2026 gave 65.0.",
       "level": "verified",
       "sources": [
        "s16",
        "s22"
       ],
       "data": {
        "model": "Gemini 3 Pro",
        "date": "2025-12",
        "avg": 70.5,
        "rl": false
       }
      },
      {
       "value": 67.5,
       "display": "Qwen3.5-397B-A17B (open weights), model card created 2026-02-16.",
       "level": "verified",
       "sources": [
        "s18"
       ],
       "data": {
        "model": "Qwen3.5-397B-A17B",
        "date": "2026-02",
        "avg": 67.5,
        "rl": false
       }
      },
      {
       "value": 58.8,
       "display": "Seed1.8 (ByteDance Seed), 2026-03.",
       "level": "verified",
       "sources": [
        "s21"
       ],
       "data": {
        "model": "Seed1.8",
        "date": "2026-03",
        "avg": 58.8,
        "rl": false
       }
      },
      {
       "value": 62.3,
       "display": "HY-Embodied-0.5 MoE-A32B (Tencent Robotics X and HY Vision Team), 2026-04.",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "data": {
        "model": "HY-Embodied-0.5 MoE-A32B",
        "date": "2026-04",
        "avg": 62.3,
        "rl": false
       }
      },
      {
       "value": 73,
       "display": "Gemini 3.6 Flash, 2026-07, Google chart. Same chart: Gemini Robotics ER 1.6 72.5.",
       "level": "verified",
       "sources": [
        "s13"
       ],
       "data": {
        "model": "Gemini 3.6 Flash",
        "date": "2026-07",
        "avg": 73,
        "rl": false
       }
      },
      {
       "value": 78.5,
       "display": "Gemini Robotics ER 2, 2026-07, Google chart. Same chart: Opus 5 67.2, GPT 5.6 Sol 43.2.",
       "level": "verified",
       "sources": [
        "s13"
       ],
       "data": {
        "model": "Gemini Robotics ER 2",
        "date": "2026-07",
        "avg": 78.5,
        "rl": false
       }
      }
     ],
     "short": "78.5% in July 2026, from Google's own chart"
    },
    "license_code": {
     "value": "CC-BY-4.0",
     "level": "verified",
     "sources": [
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "One LICENSE file (Creative Commons Attribution 4.0) at the repository root covers the scripts as well as the data. GitHub reports CC-BY-4.0.",
     "short": "CC BY 4.0"
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The same LICENSE covers data/erqa.tfrecord.",
     "short": "CC BY 4.0"
    },
    "license_assets": {
     "value": "third-party images, mixed terms",
     "display": "Some images come from five public datasets with their own terms.",
     "level": "inferred",
     "sources": [
      "s2",
      "s35",
      "s36",
      "s37",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "The report does not say which images come from which source, or on what basis third-party images are re-shared under CC BY 4.0. Not legal advice.",
     "items": [
      {
       "value": "HoloAssist: CDLA-Permissive-2.0",
       "display": "The HoloAssist site says the data is released under the CDLAv2 licence.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "Open X-Embodiment: CC BY 4.0",
       "display": "OXE README: materials other than software are under CC BY 4.0. Its component datasets were not checked.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "UMI Data: not stated",
       "display": "The UMI repository's code is MIT; its README states no data licence.",
       "level": "inferred",
       "sources": [
        "s38"
       ]
      },
      {
       "value": "MECCANO: none found",
       "level": "unknown",
       "sources": [],
       "note": "No LICENSE file in the GitHub repository (fpv-iplab/MECCANO) and no licence text on the project site."
      },
      {
       "value": "EGTEA Gaze+: not checked",
       "level": "unknown",
       "sources": [],
       "note": "The project site (cbs.ic.gatech.edu/fpv) did not load on 2026-10-10."
      }
     ],
     "short": "Third-party images with mixed terms"
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s4",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The test file is in the public GitHub repository. No registration.",
     "short": "Open. The test file is on GitHub."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s5",
      "s2",
      "s35",
      "s36",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "The ERQA licence (CC BY 4.0) allows commercial use with attribution. Some images come from third-party datasets whose terms we could not confirm (MECCANO, EGTEA Gaze+, UMI Data). Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No published study compares ERQA scores with robot task success.",
     "level": "inferred",
     "sources": [
      "s2",
      "s12",
      "s27",
      "s39",
      "s15",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "The Gemini Robotics report says zero-shot robot control 'is strongly correlated with better embodied understanding', based on one pair of models (Gemini 2.0 Flash and Gemini Robotics-ER); it does not report ERQA for Gemini Robotics-ER or relate ERQA scores to robot results. Gemini Robotics 1.5 reports ERQA (54.8 vs 47.5) and real-robot agent results for Gemini Robotics-ER 1.5 and Gemini 2.5 Flash, but ERQA is one of 15 benchmarks in its ER score and no comparison is made. Vlaser (ICLR 2026) found that gains on standard embodied-reasoning benchmarks, ERQA among them, did not carry over to robot control in simulation (SimplerEnv); that is a simulation result, not a real-robot comparison. ERIQ (arXiv 2512.24125) reports a positive correlation for its own benchmark, not for ERQA.",
     "short": "We found no study that compares it with robot task success."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Offline question set; no robot runs."
    },
    "citations": {
     "value": 482,
     "display": "482 (Semantic Scholar record of the Gemini Robotics report; 43 influential). Counts every citation of the report, not only uses of ERQA.",
     "level": "verified",
     "sources": [
      "s33"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar has two records for arXiv 2503.20020; the one keyed by arXiv ID shows 0 citations. ERQA has no paper of its own.",
     "short": "482, counting all citations of the Gemini Robotics report"
    },
    "github_stars": {
     "value": 301,
     "display": "301 stars, 16 forks (embodiedreasoning/ERQA)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "301"
    },
    "used_by": {
     "value": "At least 15 model reports published ERQA scores between 2025-03 and 2026-09 (our count of reports we opened).",
     "level": "inferred",
     "sources": [
      "s2",
      "s12",
      "s13",
      "s16",
      "s17",
      "s18",
      "s19",
      "s20",
      "s21",
      "s22",
      "s23",
      "s24",
      "s25",
      "s26",
      "s27",
      "s32",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "A lower bound: only reports we opened are counted. Toolkits that run ERQA include VLMEvalKit (used by EO-1 and InternVL3.5) and EvalScope (issue #6).",
     "items": [
      {
       "value": "Gemini Robotics / Gemini Robotics-ER",
       "display": "Google DeepMind, 2025-03. Introduces ERQA.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Gemini Robotics 1.5",
       "display": "Google DeepMind, 2025-10. One of 15 benchmarks in its embodied-reasoning score.",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "Gemini 3 Pro",
       "display": "Google, 2025-12. Vision benchmark table lists ERQA under Spatial.",
       "level": "verified",
       "sources": [
        "s16"
       ]
      },
      {
       "value": "Gemini Robotics ER 2",
       "display": "Google DeepMind, 2026-07. Chart also gives ER 1.6, Gemini 3.6 Flash, Opus 5 and GPT 5.6 Sol.",
       "level": "verified",
       "sources": [
        "s13"
       ]
      },
      {
       "value": "InternVL3.5",
       "display": "Shanghai AI Laboratory, 2025-08.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "GLM-4.5V / GLM-4.1V-Thinking",
       "display": "Zhipu AI and Tsinghua University, 2025 (GLM-4.5V 50.0).",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "EO-1",
       "display": "2025-08; work supported by Shanghai AI Laboratory (EO-1 3B: 45.5).",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "Vlaser",
       "display": "2025-10, ICLR 2026 (Vlaser-8B: 41.0).",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "Qwen3-VL",
       "display": "Qwen Team, 2025-11.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "Qwen3.5-397B-A17B",
       "display": "Model card, 2026-02.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "Xiaomi-Robotics-0",
       "display": "Xiaomi, 2026-02 (40.8).",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "Seed1.8",
       "display": "ByteDance Seed, 2026-03.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "HY-Embodied-0.5",
       "display": "Tencent Robotics X and HY Vision Team, 2026-04.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "Wall-OSS-0.5",
       "display": "X Square Robot, 2026-05 (32.8).",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Ouroboros-Spatial",
       "display": "2026-06 (Ouro-Spatial-8B: 44.0).",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "LightNav-0",
       "display": "2026-08 (LightNav-ER 4B: 43.8).",
       "level": "verified",
       "sources": [
        "s26"
       ]
      }
     ],
     "short": "At least 15 model reports, from March 2025 to September 2026"
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Alibaba (Qwen Team)",
      "ByteDance (Seed)",
      "Tencent",
      "Xiaomi",
      "X Square Robot",
      "Zhipu AI",
      "Shanghai AI Laboratory"
     ],
     "level": "verified",
     "sources": [
      "s12",
      "s13",
      "s17",
      "s21",
      "s22",
      "s23",
      "s25",
      "s20",
      "s19"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "Built ERQA and reports it for every Gemini Robotics-ER release (1.0, 1.5, 1.6, 2).",
       "level": "verified",
       "sources": [
        "s2",
        "s12",
        "s13"
       ]
      },
      {
       "value": "Alibaba (Qwen Team)",
       "display": "Qwen3-VL report and Qwen3.5 model card.",
       "level": "verified",
       "sources": [
        "s17",
        "s18"
       ]
      },
      {
       "value": "ByteDance (Seed)",
       "display": "Seed1.8 model card.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Tencent",
       "display": "HY-Embodied-0.5 report.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "Xiaomi",
       "display": "Xiaomi-Robotics-0 report, as a check that its VLA keeps vision-language skills.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "X Square Robot",
       "display": "Wall-OSS-0.5 report.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Zhipu AI",
       "display": "GLM-4.5V report (with Tsinghua University).",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Shanghai AI Laboratory",
       "display": "InternVL3.5 report.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "ERQA-Plus (A*STAR)",
      "ERQA+ (BAAI)"
     ],
     "display": "Two different benchmarks with nearly the same name build on ERQA.",
     "level": "verified",
     "sources": [
      "s30",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "Name collision: 'ERQA-Plus' (A*STAR, Singapore) and 'ERQA+' (BAAI FlagEval) are separate benchmarks.",
     "items": [
      {
       "value": "ERQA-Plus (A*STAR, Singapore)",
       "display": "2026-06. 1,766 questions on 711 images, 630 of them from ERQA. Questions generated and revised by GPT-4o, with Gemini 3 Pro as the default judge; the paper also reports a human evaluation.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "ERQA+ (BAAI FlagEval)",
       "display": "2025. 800 newly annotated questions on first-person robot video frames, built to complement ERQA with reduced contamination as a stated goal. Code listed as 'Coming Soon'.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ],
     "short": "ERQA-Plus and ERQA+, which are two separate benchmarks"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "The same model gets different scores in different reports",
     "text": "Reports use different graders, prompts and toolkits, and some copy numbers from other papers. GPT-5 scored 59.0 in Google's run (Gemini 2.5 Flash as grader) and 65.7 in InternVL3.5's run (VLMEvalKit). Gemini 3 Pro scored 70.5 in Google's table and 65.0 in HY-Embodied-0.5's API run (March 2026). Claude Opus 4.5 scored 51.3 in Google's table and 46.8 in the Qwen3.5 model card. Qwen3-VL-4B is reported at 47.3 (HY-Embodied-0.5), 41.2 (Ouroboros-Spatial), 40.0 (Xiaomi-Robotics-0) and 39.5 (LightNav-0). Google's July 2026 chart gives GPT 5.6 Sol 43.2, below the 59.0 Google reported for GPT-5 in 2025, and states no settings. Some papers also mislabel cited numbers: EO-1 lists 46.3 for Gemini 1.5 Flash, which the Gemini Robotics report gives to Gemini 2.0 Flash (1.5 Flash: 42.3), and InternVL3.5 attributes 48.3 to Gemini-2.5-Pro, which the report gives to Gemini 2.0 Pro Experimental.",
     "level": "verified",
     "sources": [
      "s12",
      "s19",
      "s16",
      "s22",
      "s18",
      "s32",
      "s23",
      "s26",
      "s13",
      "s24",
      "s2"
     ],
     "status": "open",
     "short": "The same model's score differs by up to 7.8 points between reports."
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "Grader and prompt choices move scores",
     "text": "The released script counts an answer as correct only if the entire reply equals the answer letter, so any explanation counts as wrong. Google's Gemini Robotics 1.5 report used Gemini 2.5 Flash to grade answers; Qwen3-VL, EO-1 and InternVL3.5 used their own prompts or toolkits. In the first report, a chain-of-thought instruction raised Gemini 2.0 Pro Experimental from 48.3 to 54.8 and Claude 3.5 Sonnet from 35.5 to 45.8. The chain-of-thought code was not released; a user who followed the paper got 46.50% where the report gives 50.3% (issue #3, open since 2025-07-06, no reply).",
     "level": "verified",
     "sources": [
      "s6",
      "s12",
      "s17",
      "s24",
      "s19",
      "s2",
      "s10"
     ],
     "status": "open",
     "short": "The grader (the program or model that marks answers) and the prompt change scores. A prompt that asks the model to reason step by step added up to 10.3 points."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "The test is small and many models scored close together",
     "text": "ERQA has 400 questions, and four categories have fewer than 40 (pointing 34, multi-view 37, task reasoning 38, other 14). A 95% sampling interval on 400 questions is about ±4.9 points at 50% accuracy (our calculation), so small gaps between models are within noise. MV-RoboBench (ICLR 2026) ran 11 models on ERQA and found most between 40% and 50%, with Qwen2.5-VL-7B (43.11%) less than 3 points behind GPT-4o (46.00%). Its authors judged ERQA to have low discriminative power and used another benchmark for their analysis. Later frontier results spread more widely (43.2 to 78.5 in Google's July 2026 chart).",
     "level": "verified",
     "sources": [
      "s3",
      "s28",
      "s13"
     ],
     "status": "open",
     "note": "The ±4.9 figure is our binomial calculation, not a published number.",
     "short": "With 400 questions, scores have about ±5 points of sampling noise. One study found most models scored between 40% and 50%."
    },
    {
     "id": "i4",
     "type": "contamination",
     "title": "The answers are public and some images come from robot training datasets",
     "text": "All 400 questions and answers have been public on GitHub since March 2025 under CC BY 4.0, with no hidden test split, so later models may have seen them in training. No study has measured this. ERQA's images partly come from Open X-Embodiment and other public datasets that are used to train robot models. BAAI built its separate ERQA+ benchmark from newly annotated robot videos and lists reduced contamination as a goal, contrasting it with benchmarks that reuse earlier data.",
     "level": "inferred",
     "sources": [
      "s4",
      "s5",
      "s2",
      "s31"
     ],
     "status": "open",
     "note": "A risk, not a measured leak.",
     "short": "All questions and answers are public, and there is no hidden test set. No study has measured whether models saw them during training."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "An ERQA score shows how often a vision-language model picks the right answer to questions about robot and first-person images. It is not evidence that a robot driven by that model will succeed. No study has linked ERQA scores to robot task success, and Vlaser found that gains on such benchmarks did not carry over to robot control in simulation.",
     "basis": [
      "facts.sim_to_real",
      "facts.venue"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "ERQA measures answers to questions about images. No study has linked it to robot success."
    },
    {
     "id": "r2",
     "text": "Compare ERQA numbers only within one report. Across reports, graders and prompts differ and the same model can move by up to 7.8 points. Within a report, gaps under about 5 points are inside the sampling noise of 400 questions.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "issues.i3",
      "facts.uncertainty_reported"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Compare scores only within one report. Ignore gaps of less than about 5 points."
    },
    {
     "id": "r3",
     "text": "Most of the highest published scores come from Google, which built ERQA, chose the grader and runs competitors' models itself. An independent run gave Gemini 3 Pro 65.0 where Google's table gives 70.5. Read vendor charts as the vendor's own measurement.",
     "basis": [
      "facts.top_score",
      "facts.evaluator",
      "issues.i1"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Most top scores come from Google's own runs on its own benchmark."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "Whether a robot driven by the model will succeed at its task.",
     "sub": "No study has linked ERQA scores to robot task success.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a small gap between two models is real.",
     "sub": "With only 400 questions, scores have about ±5 points of sampling noise.",
     "basis": [
      "facts.uncertainty_reported",
      "issues.i3"
     ]
    },
    {
     "id": "l3",
     "text": "How people would score on the same questions.",
     "sub": "No human baseline (a score from people taking the test) has been published.",
     "basis": [
      "facts.human_baseline"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Gemini Robotics report (2503.20020v1, full text), Gemini Robotics 1.5 report (2510.03342v3, Sections 3 and 4, Appendix C), Gemini Robotics ER 1.6 post, ER 2 post and model card, Gemini Robotics 2 post; Vlaser (2510.11027), ERIQ (2512.24125), MV-RoboBench (2510.19400), A2Eval (2602.01640); web searches on 2026-10-10 for studies that relate ERQA or embodied-reasoning QA scores to robot or VLA success. None pairs ERQA scores with real-robot results.",
     "date": "2026-10-10"
    },
    {
     "for": "human_baseline",
     "where": "Gemini Robotics report Section 2.1 and Tables 1 and 2; Gemini Robotics 1.5 Appendix C; ERQA README; ER 2 release post and model card. None reports a human score.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "Google posts (Gemini Robotics, Gemini Robotics 1.5, Gemini 3 Pro vision, ER 1.6, ER 2), the model reports under used_by, the Qwen3.5 model card, and web searches for newer or higher ERQA scores. The ER 1.6 post has no ERQA number; its 72.5 comes from the ER 2 chart. The Gemini 3 Pro model card and evaluation PDF have no ERQA row; the 70.5 comes from Google's Gemini 3 Pro vision post.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "ERQA README and LICENSE; report Section 2.1; HoloAssist site; Open X-Embodiment README; UMI repository README; MECCANO GitHub repository and project site (no licence text found); EGTEA Gaze+ site (did not load).",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "ERQA README and reports; Embodied Arena paper (2509.15273) and site (script-only page); EASI community board API (no ERQA column).",
     "date": "2026-10-10"
    },
    {
     "for": "citations",
     "where": "Semantic Scholar API by arXiv ID (record with 0 citations) and by title search (record with 482 citations, DOI 10.48550/arXiv.2503.20020).",
     "date": "2026-10-10"
    },
    {
     "for": "region",
     "where": "Report author list, README, deepmind.google careers and about pages (no office list in text).",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Gemini Robotics: Bringing AI into the Physical World (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2503.20020",
     "type": "paper",
     "publisher": "arXiv (Gemini Robotics Team, Google DeepMind)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Gemini Robotics report, full text (Section 2.1, Tables 1 and 2, Figure 4)",
     "url": "https://arxiv.org/html/2503.20020v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "ERQA question categories, Figure 4 source file with counts",
     "url": "https://arxiv.org/html/2503.20020v1/erqa_categories.svg",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "embodiedreasoning/ERQA README",
     "url": "https://github.com/embodiedreasoning/ERQA",
     "type": "repo",
     "publisher": "Google DeepMind (repository by a report author)",
     "date": "2025-03-12",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "ERQA LICENSE (Creative Commons Attribution 4.0)",
     "url": "https://github.com/embodiedreasoning/ERQA/blob/main/LICENSE",
     "type": "repo",
     "publisher": "embodiedreasoning",
     "date": "2025-03-12",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "ERQA evaluation harness (eval_harness.py)",
     "url": "https://github.com/embodiedreasoning/ERQA/blob/main/eval_harness.py",
     "type": "repo",
     "publisher": "embodiedreasoning",
     "date": "2025-03-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: embodiedreasoning/ERQA (stars, forks, created, pushed, licence)",
     "url": "https://api.github.com/repos/embodiedreasoning/ERQA",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "ERQA commit history",
     "url": "https://github.com/embodiedreasoning/ERQA/commits/main",
     "type": "repo",
     "publisher": "embodiedreasoning",
     "date": "2025-03-12",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "ERQA issues (six issues, none answered by a maintainer)",
     "url": "https://github.com/embodiedreasoning/ERQA/issues",
     "type": "repo",
     "publisher": "embodiedreasoning",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "ERQA issue #3: Evaluation with CoT",
     "url": "https://github.com/embodiedreasoning/ERQA/issues/3",
     "type": "repo",
     "publisher": "embodiedreasoning (user report)",
     "date": "2025-07-06",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Gemini Robotics brings AI into the physical world (launch post)",
     "url": "https://deepmind.google/discover/blog/gemini-robotics-brings-ai-into-the-physical-world/",
     "type": "blog",
     "publisher": "Google DeepMind",
     "date": "2025-03-12",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Gemini Robotics 1.5 report, full text v3 (Appendix C.1, Table 19)",
     "url": "https://arxiv.org/html/2510.03342v3",
     "type": "paper",
     "publisher": "arXiv (Gemini Robotics Team, Google DeepMind)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Gemini Robotics ER 2 release post (chart: ER metrics comparison)",
     "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/gemini-robotics-er-2/",
     "type": "blog",
     "publisher": "Google",
     "date": "2026-07-30",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "Gemini Robotics ER 2 model card",
     "url": "https://deepmind.google/models/model-cards/gemini-robotics-er-2/",
     "type": "site",
     "publisher": "Google DeepMind",
     "date": "2026-07-30",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Gemini Robotics-ER 1.6 post (no ERQA number)",
     "url": "https://deepmind.google/blog/gemini-robotics-er-1-6/",
     "type": "blog",
     "publisher": "Google DeepMind",
     "date": "2026-04-14",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Gemini 3 Pro: the frontier of vision AI (benchmark table image)",
     "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemini-3-pro-vision/",
     "type": "blog",
     "publisher": "Google",
     "date": "2025-12-05",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Qwen3-VL Technical Report (Section 5.8)",
     "url": "https://arxiv.org/abs/2511.21631",
     "type": "paper",
     "publisher": "arXiv (Qwen Team)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Qwen3.5-397B-A17B model card (Spatial Intelligence table)",
     "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
     "type": "repo",
     "publisher": "Qwen (Hugging Face)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "InternVL3.5 report (Table 2 and Table 11)",
     "url": "https://arxiv.org/abs/2508.18265",
     "type": "paper",
     "publisher": "arXiv (InternVL Team, Shanghai AI Laboratory)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "GLM-4.5V and GLM-4.1V-Thinking report, v6 (Table 2)",
     "url": "https://arxiv.org/abs/2507.01006",
     "type": "paper",
     "publisher": "arXiv (GLM-V Team, Zhipu AI and Tsinghua University)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Seed1.8 Model Card (Table 2)",
     "url": "https://arxiv.org/abs/2603.20633",
     "type": "paper",
     "publisher": "arXiv (ByteDance Seed)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "HY-Embodied-0.5 report (Tables 1 and 2)",
     "url": "https://arxiv.org/abs/2604.07430",
     "type": "paper",
     "publisher": "arXiv (Tencent Robotics X and HY Vision Team)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Xiaomi-Robotics-0 report (Table 3)",
     "url": "https://arxiv.org/abs/2602.12684",
     "type": "paper",
     "publisher": "arXiv (Xiaomi Robotics)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "EO-1 report, v5 (Table 2)",
     "url": "https://arxiv.org/abs/2508.21112",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Wall-OSS-0.5 Technical Report (Table 7)",
     "url": "https://arxiv.org/abs/2605.30877",
     "type": "paper",
     "publisher": "arXiv (X Square Robot)",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "LightNav-0 (Table II)",
     "url": "https://arxiv.org/abs/2608.30935",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Vlaser: Vision-Language-Action Model with Synergistic Embodied Reasoning (Table 1, Section 3.2)",
     "url": "https://arxiv.org/abs/2510.11027",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Seeing Across Views: MV-RoboBench (Appendix D.1: evaluation on ERQA)",
     "url": "https://arxiv.org/abs/2510.19400",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "A2Eval: Agentic and Automated Evaluation for Embodied Brain",
     "url": "https://arxiv.org/abs/2602.01640",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "ERQA-Plus: A Diagnostic Benchmark for Reasoning in Embodied AI",
     "url": "https://arxiv.org/abs/2606.17639",
     "type": "paper",
     "publisher": "arXiv (A*STAR, Singapore)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "ERQA+: An Enhanced Benchmark on Embodied Reasoning (project page)",
     "url": "https://flageval-baai.github.io/ERQA-Plus-page/",
     "type": "site",
     "publisher": "BAAI FlagEval Team",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Ouroboros-Spatial (Table 2: ERQA)",
     "url": "https://arxiv.org/abs/2606.11719",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Semantic Scholar API title search for the Gemini Robotics report (citation counts)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/search?query=Gemini+Robotics+Bringing+AI+into+the+Physical+World&fields=title,citationCount,influentialCitationCount,externalIds,year,publicationDate&limit=10",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Embodied Arena: A Comprehensive, Unified, and Evolving Evaluation Platform for Embodied AI",
     "url": "https://arxiv.org/abs/2509.15273",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "HoloAssist project site (data licence)",
     "url": "https://holoassist.github.io/",
     "type": "site",
     "publisher": "HoloAssist authors",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Open X-Embodiment README (licence section)",
     "url": "https://github.com/google-deepmind/open_x_embodiment",
     "type": "repo",
     "publisher": "Google DeepMind",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "MECCANO repository (no licence file)",
     "url": "https://github.com/fpv-iplab/MECCANO",
     "type": "repo",
     "publisher": "fpv-iplab",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "Universal Manipulation Interface repository (code licence, data download notes)",
     "url": "https://github.com/real-stanford/universal_manipulation_interface",
     "type": "repo",
     "publisher": "Stanford REAL lab",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "ERIQ: Unified Embodied VLM Reasoning with Robotic Action via Autoregressive Discretized Pre-training",
     "url": "https://arxiv.org/abs/2512.24125",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "Gemini Robotics 2 brings whole body intelligence to robots (announcement)",
     "url": "https://deepmind.google/blog/gemini-robotics-2-brings-whole-body-intelligence-to-robots/",
     "type": "blog",
     "publisher": "Google DeepMind",
     "date": "2026-07-30",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "Gemini Robotics 1.5 arXiv abstract page (version history)",
     "url": "https://arxiv.org/abs/2510.03342",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "GitHub API: ERQA git tree (file list and sizes)",
     "url": "https://api.github.com/repos/embodiedreasoning/ERQA/git/trees/main?recursive=1",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Basic entry created (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Full entry written from primary sources, starting from the basic entry and the frontier-labs inventory record; every fact re-checked. Added question-category counts (Figure 4 source file), scoring details from the released script, results from 15 model reports, the Google ER 2 and Gemini 3 Pro charts, issues on reporting, grading, test size and contamination, and licence terms of image sources. Corrections to earlier records: sim_to_real is now 'none-found' (was unknown); citations are 482 on the report's second Semantic Scholar record (the first shows 0); scale now lists categories. Confirmed prior leads: Gemini 2.5 Flash grading in GR 1.5, Feb 2025 results, ER 2 at 78.5%."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "eval-actions",
   "name": "Eval-Actions",
   "aliases": [
    "Eval-Actions / AutoEval",
    "AutoEval-S / AutoEval-P (its reference evaluator; not Berkeley AutoEval)",
    "Trustworthy Evaluation of Robotic Manipulation: A New Benchmark and AutoEval Methods (v1 title)",
    "TERM-Bench (repo name)",
    "EAS (annotated subset)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Scores execution quality of recorded real-robot manipulation episodes and trains/tests judge models (AutoEval) on expert grades. It yields a score for policies only via the judge; the main object scored is the evaluator. Data not yet released.",
   "summary": {
    "text": "Real-robot episodes with expert quality grades, used to train and test models that judge how well manipulation was executed.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No affiliations in arXiv v1 or v2 HTML; v2 and project page say 'Anonymous'. Repo owner is GitHub user LogSSim."
    },
    "builder_type": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Affiliations not stated."
    },
    "region": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Affiliations not stated."
    },
    "first_release": {
     "value": "2026-01 (arXiv v1 2026-01-26)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "v1 title: 'Trustworthy Evaluation of Robotic Manipulation: A New Benchmark and AutoEval Methods' (https://arxiv.org/abs/2601.18723v1)."
    },
    "latest_update": {
     "value": "arXiv v2 2026-06-28 (new title; reframed as diagnostic methodology; abstract adds 13K+ episodes / 150+ tasks / 52 h)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Repo last commit 2026-01-27 (README update)."
    },
    "version": {
     "value": "arXiv v2; dataset 'coming soon'; code repo initial commits only",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Project page: 'Code (Coming Soon)', 'Dataset (Coming Soon)'. Repo README: 'Dataset coming soon', download links 'coming soon'."
    },
    "published_at": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "arXiv comments give only site and code links. Project page links a file named 'T_RO.pdf'; we did not treat that as evidence of a venue."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Judges score recorded episodes; no control loop at evaluation time."
    },
    "robots": {
     "value": "ARX R5 and UR5 manipulators (single-arm and bimanual)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Joint trajectories 7/14-DoF."
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Examples: grasping, bowl stacking, table cleaning, medicine-box organisation, plate handover."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Task examples are tabletop; scene types not enumerated."
    },
    "scale": {
     "value": "13K+ episodes; 150+ tasks; about 52 h; 2.8K failures. EAS subset: 6K+ episodes, 50+ tasks, 12 h, 37.4% failure ratio, split 80/10/10",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Table III; project page repeats it."
    },
    "scoring": {
     "value": [
      "human-rating",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Expert Grading: 10 experts score 1-10 on success, collisions, smoothness, efficiency. Rank-Guided labels: physical indicators weighted to match expert rankings (fit on train split). CoT text explanations. Evaluators are scored by SRCC to these labels and success accuracy."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on page, paper or repo."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file is Apache 2.0, but the README shows an MIT badge and a commented-out 'released under the MIT License' line: conflict, LICENSE file taken as authoritative."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Dataset not released; checked project page, repo README and paper."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Data are real-robot recordings scored offline, so sim-to-real does not apply. Validity evidence is judge-vs-expert agreement: AutoEval-S SRCC 0.81 (EG) and 0.84 (RG), success accuracy 90.6% / 91.0%; AutoEval-P SRCC 0.70 (CoT); inter-expert leave-one-out SRCC 0.91 +/- 0.06, ICC(2,1) 0.88. Policy-level use: 5 policies x 5 tasks x 20 rollouts; AutoEval-EG ranking differs from success-rate ranking (RDT vs ACT). All measured by the authors."
    },
    "human_agreement": {
     "value": "AutoEval-S SRCC 0.81 (EG), 0.84 (RG); success accuracy 90.6% / 91.0%; AutoEval-P SRCC 0.70 (CoT); source classification 99.6%",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "On the EAS test split; authors' own measurement."
    },
    "citations": {
     "value": 5,
     "display": "5 (Semantic Scholar; 1 influential)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "2026-10-10."
    },
    "github_stars": {
     "value": 11,
     "display": "11 (LogSSim/TERM-Bench)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "2026-10-10."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper revised 2026-06-28; data release still pending."
    },
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Limited scope of quality score",
     "text": "Authors: scores mainly reflect spatial generalisation, not language/object/long-horizon generalisation; collision/safety assessment incomplete; judge may fail under occlusion.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Eval-Actions: Fine-Grained Execution Quality Evaluation for Robotic Manipulation (full text)",
     "url": "https://arxiv.org/html/2601.18723v2",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s2": {
     "title": "Eval-Actions: Fine-Grained Execution Quality Evaluation for Robotic Manipulation",
     "url": "https://arxiv.org/abs/2601.18723",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s3": {
     "title": "Eval-Actions: Fine-Grained Execution Quality Evaluation for Robotic Manipulation",
     "url": "https://eval-actions.github.io/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "LogSSim/TERM-Bench on GitHub (blob)",
     "url": "https://github.com/LogSSim/TERM-Bench/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s6": {
     "title": "LogSSim/TERM-Bench on GitHub (repository)",
     "url": "https://api.github.com/repos/LogSSim/TERM-Bench",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "ewmbench",
   "name": "EWMBench",
   "full_name": "EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models",
   "aliases": [
    "Embodied World Model Benchmark",
    "EWMBM",
    "WMBM"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "World-model evaluation built for robot manipulation: it scores generated robot videos against real AgiBot World episodes.",
   "summary": {
    "text": "EWMBench scores AI video generators (world models) on how closely their videos of a robot doing a task match recorded videos of the real task, for scene stability, gripper motion and instruction meaning. AgiBot built it from 10 tasks in its AgiBot World data, and automatic tools, including a vision-language model, do the scoring.",
    "short": "EWMBench is a benchmark for world models, which here are AI models that generate videos of a robot doing a task. It scores how closely those videos match real recordings of the same task.",
    "sources": [
     "s1",
     "s6",
     "s15"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents a curated dataset, fixed metrics and an open evaluation toolkit, and calls it a benchmark."
    },
    "publishers": {
     "value": [
      "AgiBot",
      "Shanghai Jiao Tong University",
      "Harbin Institute of Technology",
      "National University of Singapore / CUHK MMLab"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Yue Hu, Siyuan Huang, Yue Liao, Shengcong Chen, Pengfei Zhou, Liliang Chen, Maoqing Yao, Guanghui Ren. arXiv v1 prints the first name as 'Hu Yue'.",
     "items": [
      {
       "value": "AgiBot",
       "display": "Six of eight authors list AgiBot. The BMVC PDF says the three authors with other affiliations did the work while employed at AgiBot.",
       "level": "verified",
       "sources": [
        "s3",
        "s4"
       ]
      },
      {
       "value": "Shanghai Jiao Tong University",
       "display": "Affiliation of Siyuan Huang.",
       "level": "verified",
       "sources": [
        "s1",
        "s3"
       ]
      },
      {
       "value": "Harbin Institute of Technology (Weihai)",
       "display": "Second affiliation of Yue Hu.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "MMLab-CUHK or National University of Singapore",
       "display": "Affiliation of Yue Liao. CONFLICT: arXiv says MMLab-CUHK; the BMVC page and PDF say National University of Singapore.",
       "level": "verified",
       "sources": [
        "s1",
        "s3",
        "s4"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "robot-company",
     "level": "inferred",
     "sources": [
      "s3",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Six of eight authors list AgiBot, and the BMVC PDF says the other three did the work as AgiBot employees."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Lead organisation AgiBot; the BMVC PDF gives its address in Shanghai. One author lists a Singapore or Hong Kong affiliation."
    },
    "first_release": {
     "value": "2025-05",
     "display": "arXiv v1 on 2025-05-14. GitHub repository created 2025-05-14; Hugging Face data created 2025-05-15. Published at BMVC 2025 (Sheffield, 24 to 27 November 2025).",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s16",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "The BMVC page header says '35th' conference while its citation block says '36th'.",
     "short": "May 2025, at BMVC 2025"
    },
    "latest_update": {
     "value": "2025-06",
     "display": "Last code commit 2025-06-13. The Hugging Face data has not changed since 2025-05-16.",
     "level": "verified",
     "sources": [
      "s7",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "The commit 'add psnr and ssim' is dated 2025-06-11. The full ground-truth set promised in issue #1 has not appeared (see issues.i1).",
     "short": "June 2025. The code has not changed since then."
    },
    "version": {
     "value": "arXiv v2",
     "display": "No tags or releases. Paper arXiv v2 (2025-05-18). Evaluation model weights are labelled v0.1.",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s18",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "We compared the arXiv v1 and v2 texts: they differ in author-name order and citation style, and every decimal number is the same. The BMVC camera-ready has the same Table 2.",
     "short": "No releases. The paper is at version 2."
    },
    "status": {
     "value": "dormant",
     "display": "No code or data change since June 2025. AgiBot still uses three of its metrics in its 2026 contest.",
     "level": "inferred",
     "sources": [
      "s7",
      "s14",
      "s39",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Last commit 2025-06-13. Users asked about the missing full test set in issue #1 on 2025-07-09, 2025-10-16 and 2025-11-04 without a reply; issue #2 (2025-11-10) is unanswered. The basic entry said 'maintained'; changed because nothing has been updated for over a year.",
     "short": "No code or data changes since June 2025. Its metrics are still reused."
    },
    "capability": {
     "value": [
      "world-modeling",
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "The models under test generate videos of robot manipulation from a first frame and an instruction."
    },
    "generalisation": {
     "value": [
      "none-stated"
     ],
     "display": "The paper does not describe held-out conditions. Test episodes were chosen to have varied arm paths.",
     "level": "inferred",
     "sources": [
      "s1",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Episodes were picked by a greedy rule that maximises differences between voxelised gripper paths. Genie Envisioner says the 10 tasks were left out of GE-Base pre-training; the EWMBench paper does not say whether other tested models saw these tasks. The two top models are described as fine-tuned for embodied scenes, without naming the data.",
     "short": "Not described in the paper"
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Generated videos are compared with recorded real-robot videos. No robot moves and no policy runs."
    },
    "embodiment": {
     "value": [
      "bimanual-arm",
      "humanoid"
     ],
     "display": "Videos show a two-armed robot from its head camera. The model under test is a video generator.",
     "level": "inferred",
     "sources": [
      "s1",
      "s35"
     ],
     "checked": "2026-10-10",
     "note": "The EWMBench paper does not name the robot; its trajectory metrics track left and right end effectors. The source data, AgiBot World, was collected with 'dual-arm humanoid robots' on mobile bases (AgiBot World paper).",
     "short": "The videos show a two-armed robot."
    },
    "scene": {
     "value": [
      "kitchen",
      "home",
      "retail-logistics",
      "industrial"
     ],
     "display": "Household, commercial and industrial settings, per the paper.",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Mapped by us from the task list: toaster, pouring water, cutlery, microwave (kitchen); showerhead, drawer, bottle cleaning, ice (home); freezer restocking (retail); detergent packing (industrial)."
    },
    "tasks": {
     "value": 10,
     "display": "10 manipulation tasks from AgiBot World, each split into 4 to 10 captioned sub-actions",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Tasks (Appendix A.2.1): retrieving toast from a toaster, pouring water, setting cutlery, restocking a freezer, producing ice, packing laundry detergent, cleaning bottles, heating food in a microwave, installing a showerhead, storing objects in a drawer.",
     "short": "10 tasks"
    },
    "test_set": {
     "value": 100,
     "display": "Paper: 10 episodes per task, 100 in total. Public download: 21 samples from 7 task categories.",
     "level": "verified",
     "sources": [
      "s1",
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "100 episodes (paper, Section 4)",
       "display": "10 per task, chosen from 100 sampled clips per task by a greedy diversity rule on voxelised gripper paths.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      },
      {
       "value": "30 samples (paper introduction)",
       "display": "The introduction says '30 candidate samples across ten tasks'. This conflicts with Section 4.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      },
      {
       "value": "21 samples, 7 tasks (public)",
       "display": "On 2025-06-20 a repository collaborator called the public ground truth a small verification subset and said the full version would follow. Nothing was added by 2026-10-10.",
       "level": "verified",
       "sources": [
        "s14",
        "s16"
       ]
      },
      {
       "value": "267,683,840 bytes",
       "display": "Two files on Hugging Face: gt_dataset.tar (128,450,560 bytes) and generated_samples.tar (139,233,280 bytes, example generations).",
       "level": "verified",
       "sources": [
        "s15",
        "s17"
       ],
       "note": "The Hub viewer's '4,644 rows' are single video frames from generated_samples.tar, not episodes (datasets-server info)."
      }
     ],
     "note": "The basic entry's '4,644 rows' described frames, not test episodes; corrected here.",
     "short": "100 episodes in the paper. Only 21 samples are public."
    },
    "scoring": {
     "value": [
      "fidelity",
      "auto-judge",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Fidelity: comparison with the recorded video and gripper path. Automatic judge: Qwen2.5-VL captions and logic check. Composite: the Overall sum."
    },
    "metric_detail": {
     "value": "8 metrics summed",
     "display": "Scene (1 metric): similarity of fine-tuned DINOv2 features between frames. Motion (3): Hausdorff distance, normalised dynamic time warping (nDTW) and a velocity and acceleration match, all on gripper paths found by a fine-tuned YOLO-World detector. Semantics (4): BLEU and CLIP similarity of captions written by Qwen2.5-VL-7B, a yes-or-no logic check by the same model, and diversity across generations. Each metric is scaled to 0 to 1. The Overall score is their sum, so the maximum is 8.",
     "level": "inferred",
     "sources": [
      "s1",
      "s9",
      "s10",
      "s11",
      "s12",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Metric definitions are verified in the paper and code. That Overall is a sum is our arithmetic: the 'Avg.' columns in Table 2 are sums (EnerVerse_FT 0.9427 + 1.6676 + 2.0907 = 4.7010). In the released code the motion scores are divided by fixed constants (24.979, 22.519, 50.202) and capped at 1; the paper does not give these constants. Gripper paths are 2D image positions. The logic prompt tells the judge to count any human hand in the video as a violation.",
     "short": "The sum of 8 automatic scores, with a maximum of 8"
    },
    "trials": {
     "value": "best of 3 per episode",
     "display": "Each model makes 3 videos per episode. Only the video whose gripper path is closest to the recording (by Hausdorff distance) is scored. 2,100 videos for 7 models.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Released code",
       "display": "The summary script averages over all generated videos. We found no best-of-3 selection step.",
       "level": "inferred",
       "sources": [
        "s8",
        "s9"
       ],
       "note": "Our reading of EWMBench/__init__.py and trajectory_consistency.py."
      },
      {
       "value": "Logics values fit 90 videos",
       "display": "Every Logics value in Table 2 is a whole multiple of 1/90 (for example 0.9778 = 88/90). That fits 30 samples times 3 generations and does not fit 100 or 300 videos.",
       "level": "inferred",
       "sources": [
        "s1"
       ],
       "note": "Our arithmetic on the seven Logics values (88, 85, 87, 82, 66, 74 and 66 out of 90)."
      },
      {
       "value": "Genie Envisioner",
       "display": "Three samples per instruction; the one with the lowest Hausdorff distance is kept.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Qwen-RobotWorld",
       "display": "Evaluates on the public set of 21 samples across 7 tasks.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      }
     ],
     "short": "Best of 3 videos per episode"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s1",
      "s21",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "Table 2 gives single numbers. Genie Envisioner and Qwen-RobotWorld also give single numbers without error bars."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Users run the toolkit on their own generations. AgiBot's contest tracks run an organiser test server, but only with three of the metrics."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s6",
      "s15",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard in the repository, dataset card or paper. AgiBot's challenge leaderboards use three EWMBench metrics on their own test data (see facts.derived_benchmarks)."
    },
    "top_score": {
     "value": 4.701,
     "display": "4.7010 Overall out of 8 (EnerVerse_FT, May 2025, paper test set). The best result on the public 21-sample set is 4.60 (Qwen-RobotWorld, June 2026). The two sets differ, so the rows below are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s1",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "Each score is verified at its source. 'Best' is our judgement over the papers we found. 2025-05 rows: 100 episodes, best of 3 videos, run by the EWMBench authors. 2026-06 rows: 21 samples, run by the Qwen team; its Cosmos row is identical to the paper's COSMOS row (see issues.i2).",
     "items": [
      {
       "value": 4.701,
       "display": "EnerVerse_FT, 2025-05: Scene 0.9427, Motion 1.6676, Semantics 2.0907",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "note": "Genie Envisioner (2025-08) lists these same four numbers for GE-Base (see issues.i2).",
       "data": {
        "model": "EnerVerse_FT",
        "date": "2025-05",
        "avg": 4.701,
        "rl": false
       }
      },
      {
       "value": 4.5493,
       "display": "LTX_FT (fine-tuned LTX-Video), 2025-05",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "LTX_FT",
        "date": "2025-05",
        "avg": 4.5493,
        "rl": false
       }
      },
      {
       "value": 3.8698,
       "display": "Kling-1.6, 2025-05. Best general-purpose model in the paper.",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "Kling-1.6",
        "date": "2025-05",
        "avg": 3.8698,
        "rl": false
       }
      },
      {
       "value": 3.4125,
       "display": "Hailuo I2V-01-live, 2025-05",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "Hailuo I2V-01-live",
        "date": "2025-05",
        "avg": 3.4125,
        "rl": false
       }
      },
      {
       "value": 3.2872,
       "display": "COSMOS-7B, 2025-05",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "COSMOS-7B",
        "date": "2025-05",
        "avg": 3.2872,
        "rl": false
       }
      },
      {
       "value": 3.1392,
       "display": "OpenSora 2.0, 2025-05",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "OpenSora 2.0",
        "date": "2025-05",
        "avg": 3.1392,
        "rl": false
       }
      },
      {
       "value": 2.9676,
       "display": "LTX-Video (not fine-tuned), 2025-05",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "data": {
        "model": "LTX-Video",
        "date": "2025-05",
        "avg": 2.9676,
        "rl": false
       }
      },
      {
       "value": 4.6,
       "display": "Qwen-RobotWorld, 2026-06, public 21-sample set: Scene 0.9142, HSD 0.5660, Logics 1.0000",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "Qwen-RobotWorld",
        "date": "2026-06",
        "avg": 4.6,
        "rl": false
       }
      },
      {
       "value": 4.05,
       "display": "LVP, 2026-06, run by the Qwen team on the 21-sample set",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "LVP",
        "date": "2026-06",
        "avg": 4.05,
        "rl": false
       }
      },
      {
       "value": 3.89,
       "display": "Sora2, 2026-06, run by the Qwen team on the 21-sample set",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "Sora2",
        "date": "2026-06",
        "avg": 3.89,
        "rl": false
       }
      },
      {
       "value": 3.85,
       "display": "Kling 2.6, 2026-06, run by the Qwen team on the 21-sample set",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "Kling 2.6",
        "date": "2026-06",
        "avg": 3.85,
        "rl": false
       }
      },
      {
       "value": 3.49,
       "display": "Veo3, 2026-06, run by the Qwen team on the 21-sample set",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "Veo3",
        "date": "2026-06",
        "avg": 3.49,
        "rl": false
       }
      }
     ],
     "short": "4.70 out of 8, in May 2025",
     "chart": {
      "max": 8,
      "unit": "",
      "label": "Overall score, the sum of 8 metrics, out of 8"
     }
    },
    "license_code": {
     "value": "CC-BY-NC-SA-4.0",
     "level": "verified",
     "sources": [
      "s6",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "README: all data and code in the repository are under CC BY-NC-SA 4.0. There is no LICENSE file (HTTP 404) and GitHub detects no licence."
    },
    "license_data": {
     "value": "CC-BY-NC-SA-4.0",
     "level": "verified",
     "sources": [
      "s15",
      "s36"
     ],
     "checked": "2026-10-10",
     "note": "Hugging Face dataset card metadata; not gated. The ground-truth clips come from AgiBot World, whose README also states CC BY-NC-SA 4.0."
    },
    "license_assets": {
     "value": "CC-BY-NC-SA-4.0",
     "display": "The evaluation model weights (fine-tuned DINOv2 and YOLO-World) are CC BY-NC-SA 4.0 on Hugging Face.",
     "level": "verified",
     "sources": [
      "s18",
      "s1",
      "s13",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The detector is fine-tuned from Ultralytics yolov8s-worldv2 (paper A.1), and the toolkit imports the ultralytics package, whose LICENSE file is AGPL-3.0. The project does not say how the two licences combine. Not legal advice."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s6",
      "s15",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub; data and weights on Hugging Face without a gate or registration. Only the 21-sample subset of the ground truth is public (issues.i1).",
     "short": "Open, but only a subset of the test set is public."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s6",
      "s15",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Code, data and weights all carry the NC (non-commercial) clause of CC BY-NC-SA 4.0. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No study links EWMBench scores to how robots perform. The authors compared its model ranking with human judges (see validity).",
     "level": "inferred",
     "sources": [
      "s1",
      "s21",
      "s28"
     ],
     "checked": "2026-10-10",
     "note": "Ground truth is recorded real-robot video, but no paper tests whether a better EWMBench score means a more useful world model for policy training or evaluation. See searched.",
     "short": "No study has checked it against robot results."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Offline scoring of generated video; no robot runs."
    },
    "citations": {
     "value": 41,
     "display": "41 (Semantic Scholar; 7 influential)",
     "level": "verified",
     "sources": [
      "s20"
     ],
     "checked": "2026-10-10",
     "short": "41"
    },
    "github_stars": {
     "value": 132,
     "display": "132 stars, 10 forks (AgibotTech/EWMBench)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "132"
    },
    "dataset_downloads": {
     "value": 107,
     "display": "107 (Hub 'downloads' field), 3,924 all time, 2 likes: agibot-world/EWMBench",
     "level": "verified",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "The model-weights repository shows 0 downloads in the same API field and 4 likes.",
     "short": "107 on Hugging Face"
    },
    "used_by": {
     "value": "At least 7 papers or contests report EWMBench or EWMBench-style scores by 2026-10. Our count from Semantic Scholar's 41 citing papers; not exhaustive.",
     "level": "inferred",
     "sources": [
      "s20",
      "s21",
      "s25",
      "s26",
      "s27",
      "s32",
      "s33",
      "s34"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Genie Envisioner",
       "display": "AgiBot, 2025-08. Scores GE-Base and six other video models (Figure 17) and GE-Sim in an action-conditioned setting (Table 2).",
       "level": "verified",
       "sources": [
        "s21",
        "s22"
       ]
      },
      {
       "value": "AgiBot World Challenge @ IROS 2025, World Model track",
       "display": "Local evaluation uses the EWMBench toolkit; the online score uses PSNR, scene consistency and nDTW.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "AgiBot World Challenge @ ICRA 2026, World Model track",
       "display": "Overall is the mean of scene consistency, nDTW and PSNR clipped to 0 to 35 and divided by 35. Contest ran 2026-02-28 to 2026-04-20.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "Qwen-RobotWorld",
       "display": "Qwen team, 2026-06. Reports 4.60 on the public 21-sample set, ahead of LVP (4.05).",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Pelican-Sim 1.0",
       "display": "2026-09. Uses 'adapted EWMBench' metrics on other test sets (AgiBotWorld Beta, RoboMIND, RoboTwin).",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "EVEWorld",
       "display": "2026-10. Reports results on 'AgiBot–EWMBench'.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "ConfAL-WM",
       "display": "2026-08. Follows an 'EWMBench-style' evaluation protocol.",
       "level": "verified",
       "sources": [
        "s34"
       ]
      }
     ],
     "short": "At least 7 papers or contests by October 2026"
    },
    "industry_use": {
     "value": [
      "AgiBot",
      "Qwen team"
     ],
     "level": "verified",
     "sources": [
      "s21",
      "s26",
      "s27",
      "s25"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "AgiBot",
       "display": "Builder. Uses it in Genie Envisioner and in its IROS 2025 and ICRA 2026 world-model contests.",
       "level": "verified",
       "sources": [
        "s21",
        "s26",
        "s27"
       ]
      },
      {
       "value": "Qwen team",
       "display": "Reports EWMBench results in the Qwen-RobotWorld technical report (2026-06).",
       "level": "verified",
       "sources": [
        "s25"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "AgiBot World Challenge, World Model track"
     ],
     "display": "AgiBot's contest track scores PSNR, scene consistency and nDTW from EWMBench on its own test data (separate Atlas entry: agibot-world-challenge-wm).",
     "level": "verified",
     "sources": [
      "s26",
      "s27"
     ],
     "checked": "2026-10-10",
     "short": "AgiBot's world-model contest uses 3 of its metrics."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Most of the test set is not public",
     "text": "The paper scores models on 100 episodes from 10 tasks. The public ground truth holds 21 samples from 7 task categories. A repository collaborator called it a verification subset on 2025-06-20 and said the full version would follow. Users asked again in July, October and November 2025 without a reply, and the Hugging Face data has not changed since 2025-05-16. Later papers such as Qwen-RobotWorld score models on the 21-sample set, so their numbers are not on the same footing as the paper's.",
     "level": "verified",
     "sources": [
      "s14",
      "s16",
      "s25",
      "s1"
     ],
     "status": "open",
     "short": "Only 21 of the paper's 100 test episodes are public."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same scores appear under different model names",
     "text": "EWMBench Table 2 credits EnerVerse_FT with Scene 0.9427, Motion 1.6676, Semantics 2.0907 and Overall 4.7010. Genie Envisioner, a later AgiBot paper whose authors include all eight EWMBench authors, lists the same four numbers for GE-Base (Figure 17, unchanged from its v1 of 2025-08 to v3 of 2025-11). In EWMBench's human study the top model is LTX_FT; Genie Envisioner shows the same EWMBench and VBench bars with GE-Base in that place. Neither paper explains this. Genie Envisioner says GE-Base is built on LTX-Video. Separately, the Qwen-RobotWorld report evaluates on the 21-sample set, yet its Cosmos row is identical to the paper's COSMOS row (0.7963 to 0.7333 on all eight metrics), which came from the 100-episode protocol.",
     "level": "verified",
     "sources": [
      "s1",
      "s21",
      "s22",
      "s23",
      "s24",
      "s37",
      "s25"
     ],
     "status": "open",
     "note": "That the Qwen Cosmos row was copied rather than re-run is our inference from eight identical four-decimal values.",
     "short": "A later AgiBot paper lists EnerVerse_FT's exact scores under another model's name."
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "The released code differs from the paper's method",
     "text": "In our reading of the released toolkit, the summary script averages over all generated videos and has no best-of-3 selection. BLEU and CLIP are both computed on the one-line video summary, while the paper describes CLIP on step-by-step descriptions. Motion scores are divided by fixed constants that the paper does not report. A variable-name typo (trail_id for trial_id) attaches each Logics value to the wrong video or drops it from the summary table. Genie Envisioner also describes the ground-truth paths as manually annotated, while the EWMBench toolkit detects them automatically. Scores from the public code may therefore differ from Table 2.",
     "level": "inferred",
     "sources": [
      "s8",
      "s9",
      "s10",
      "s11",
      "s21",
      "s1",
      "s6"
     ],
     "status": "open",
     "note": "Code read at the last commit (3a5531c, 2025-06-13). We did not run it.",
     "short": "The public scoring code differs from the method the paper describes."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The human check is small and gives no statistic",
     "text": "Annotators ranked videos from four models (LTX_FT, Kling-1.6, Hailuo I2V-01-live, OpenSora-2.0), giving 3, 2 and 0 points to the best, second-best and worst. The paper gives no number of annotators or videos and no agreement statistic. The aggregated human order matched EWMBench's order for all four models, while VBench put Hailuo first. The EWMBench scores in the same figure (5.49, 4.75, 4.30, 4.03) differ from Table 2's Overall scores for these models (4.5493, 3.8698, 3.4125, 3.1392), and the paper does not explain the difference. The automatic judges (Qwen2.5-VL-7B for captions and logic) are not checked against human labels; only the gripper detector is tested (recall 0.91667, precision 1.0 on held-out frames).",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s24"
     ],
     "status": "open",
     "short": "The check against human rankings covers only four models. The paper does not say how many people took part and gives no statistic."
    },
    {
     "id": "i5",
     "type": "shortcut",
     "title": "A still video can score well on scene stability",
     "text": "The paper notes that visually plausible but static videos may score high on scene consistency while lacking meaningful motion. The Overall score adds the scene score to the motion and semantic scores, so part of the total does not depend on doing the task. OpenSora and LTX, which the paper says often produce static videos, scored 0.9210 and 0.9156 on scene consistency, above Kling (0.8888).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open",
     "short": "The paper itself notes that videos with no motion can get high scene scores."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "It does not test whether a world model helps a robot",
     "text": "Later benchmark papers say EWMBench measures the generated video and does not test whether a world model helps a robot decide or act. WorldArena 2.0 (2026-05) says such benchmarks 'do not assess whether generated dynamics support embodied decision-making'. Wow, wo, val! (2026-01) says EWMBench does not assess planning and execution. WorldArena (2026-02) marks it as lacking data-engine, policy-evaluation and action-planner tests. WorldArena also found that its own video-quality score correlated with downstream robot tasks at only r 0.600 and r 0.360, which suggests video scores and usefulness can diverge; that finding is about WorldArena's score, not EWMBench's.",
     "level": "verified",
     "sources": [
      "s29",
      "s30",
      "s28",
      "s31"
     ],
     "status": "open",
     "short": "Other research groups note that it does not test whether a world model helps a robot decide or act."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "EWMBench shows how closely a model's video of a robot task matches a recording, as judged by automatic tools. It does not show whether the model would help a robot succeed, and no study has linked its scores to robot results.",
     "basis": [
      "facts.sim_to_real",
      "facts.metric_detail",
      "issues.i6"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "EWMBench measures how much the videos look like real recordings. It does not measure robot success."
    },
    {
     "id": "r2",
     "text": "Treat published EWMBench numbers as rough guides. The full test set is not public, later papers use a 21-sample subset or adapted metrics, the released code differs from the paper, and the builder's own papers attach the same numbers to different model names.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "issues.i3",
      "facts.top_score"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Published numbers are hard to compare or reproduce."
    },
    {
     "id": "r3",
     "text": "The human check supports the ranking of four models. It is too small to show that the automatic judges are reliable across models, and its figure values differ from the main table.",
     "basis": [
      "issues.i4"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The human check is too small to show that the automatic judges are reliable."
    },
    {
     "id": "r4",
     "text": "Its metrics are still used, mainly to score AgiBot's world-model contests, which use three of the eight. The non-commercial licence limits company use outside research.",
     "basis": [
      "facts.derived_benchmarks",
      "facts.commercial_use",
      "facts.status"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "It is used mostly through AgiBot's contests. Its licence is non-commercial."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "Whether the world model helps a robot do its task.",
     "sub": "EWMBench scores how much the videos look like the recordings. It does not measure robot success.",
     "basis": [
      "facts.sim_to_real",
      "issues.i6"
     ]
    },
    {
     "id": "l2",
     "text": "How a model does on the full test set.",
     "sub": "Only 21 of the 100 test episodes are public.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "Whether the automatic judges are reliable.",
     "sub": "Only one small human check has been done, and it reports no statistic.",
     "basis": [
      "issues.i4"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "Human ranking study (EWMBench paper)",
     "date": "2025-05",
     "by": "authors",
     "method": "This was not a real-robot comparison. People ranked videos from 4 video models, and their combined order was compared with the orders from EWMBench and VBench.",
     "result": "The order was the same for all 4 models. No statistic was reported.",
     "authors_view": "align more closely with human judgments than VBench",
     "n_policies": 4,
     "level": "verified",
     "sources": [
      "s1",
      "s24",
      "s4"
     ],
     "note": "Annotator and video counts are not given. Genie Envisioner (2025-08) shows the same comparison with GE-Base in place of LTX_FT (issues.i2)."
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "EWMBench arXiv v1 and v2 and BMVC camera-ready (full text); GitHub README and issues; Hugging Face cards; Genie Envisioner v3 (2508.05635); Qwen-RobotWorld (2606.17030); WorldArena (2602.08971); WorldArena 2.0 (2605.17912); Wow, wo, val! (2601.04137); decision-making position paper (2606.15032); Pelican-Sim 1.0 (2609.12036); all 41 Semantic Scholar citing papers scanned by title, 21 opened and searched for 'EWMBench'; web searches for EWMBench critiques and correlation with policy success. No paired comparison with robot results found.",
     "date": "2026-10-10"
    },
    {
     "for": "validity (agreement with humans)",
     "where": "Paper Section 4.2 and Figure 6 (arXiv v2 and BMVC PDF, figure read as an image); Genie Envisioner Figure 18. No annotator count, video count or agreement statistic in either.",
     "date": "2026-10-10"
    },
    {
     "for": "test_set (full ground truth)",
     "where": "Hugging Face dataset tree and commit list (last change 2025-05-16), datasets-server row keys, GitHub issue #1 and #2, README. Only the 21-sample subset is available.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Paper, GitHub README, Hugging Face dataset and model cards, AgiBot ICRA 2026 world-model test server page.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code",
     "where": "GitHub contents API for LICENSE (404), repository licence field (null), README licence section.",
     "date": "2026-10-10"
    },
    {
     "for": "issues.i2 (explanation of GE-Base label)",
     "where": "Genie Envisioner v1 and v3 text and figures (figure files are byte-identical across versions); EWMBench paper. Neither explains the relabelling.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models (full text, arXiv v2)",
     "url": "https://arxiv.org/html/2505.09694v2",
     "type": "paper",
     "publisher": "arXiv (AgiBot, SJTU, CUHK MMLab, HIT)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "EWMBench arXiv abstract page (submission history)",
     "url": "https://arxiv.org/abs/2505.09694",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "EWMBench, BMVC 2025 proceedings page",
     "url": "https://bmvc2025.bmva.org/proceedings/736/",
     "type": "paper",
     "publisher": "British Machine Vision Association",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "EWMBench, BMVC 2025 camera-ready PDF",
     "url": "https://bmva-archive.org.uk/bmvc/2025/assets/papers/Paper_736/paper.pdf",
     "type": "paper",
     "publisher": "British Machine Vision Association",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "GitHub API: AgibotTech/EWMBench (stars, forks, created, pushed, licence)",
     "url": "https://api.github.com/repos/AgibotTech/EWMBench",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "EWMBench GitHub README",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/README.md",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "EWMBench commit history",
     "url": "https://github.com/AgibotTech/EWMBench/commits/main",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06-13",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "EWMBench scoring code: EWMBench/__init__.py (result merging and means)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/EWMBench/__init__.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06-13",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "EWMBench scoring code: trajectory_consistency.py (HSD, nDTW, DYN, normalisation constants)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/EWMBench/trajectory_consistency.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "EWMBench scoring code: caption.py (Qwen2.5-VL prompt and logic check)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/EWMBench/caption.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "EWMBench scoring code: semantics.py (BLEU and CLIP on captions)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/EWMBench/semantics.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "EWMBench scoring code: scene_consistency.py (DINOv2 frame similarity)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/EWMBench/scene_consistency.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "EWMBench preprocessing: processing/detection_tracking.py (ultralytics YOLO, 2D gripper paths)",
     "url": "https://github.com/AgibotTech/EWMBench/blob/main/processing/detection_tracking.py",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "EWMBench GitHub issue #1: Missing samples (collaborator reply 2025-06-20)",
     "url": "https://github.com/AgibotTech/EWMBench/issues/1",
     "type": "repo",
     "publisher": "AgiBot (repository collaborator) and users",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "agibot-world/EWMBench dataset card and file listing (Hugging Face)",
     "url": "https://huggingface.co/datasets/agibot-world/EWMBench",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Hugging Face Hub API record for agibot-world/EWMBench (downloads, likes, created, last modified, commits)",
     "url": "https://huggingface.co/api/datasets/agibot-world/EWMBench?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=lastModified&expand[]=createdAt",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Hugging Face datasets-server info for agibot-world/EWMBench (4,644 jpg rows from generated_samples.tar)",
     "url": "https://datasets-server.huggingface.co/info?dataset=agibot-world/EWMBench",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "agibot-world/EWMBench-model (Hugging Face model card: fine-tuned DINOv2 and YOLO-World weights)",
     "url": "https://huggingface.co/agibot-world/EWMBench-model",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Ultralytics LICENSE (AGPL-3.0)",
     "url": "https://github.com/ultralytics/ultralytics/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Ultralytics",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Semantic Scholar API record for arXiv:2505.09694",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2505.09694?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation (v3; EWMBench sections)",
     "url": "https://arxiv.org/html/2508.05635v3",
     "type": "paper",
     "publisher": "arXiv (AgiBot Genie Team)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Genie Envisioner Figure 17 (EWMBench results table naming GE-Base)",
     "url": "https://arxiv.org/html/2508.05635v3/Exp-8metric-v2.png",
     "type": "paper",
     "publisher": "arXiv (AgiBot Genie Team)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Genie Envisioner Figure 18 (human, VBench and EWMBench ranks with GE-Base)",
     "url": "https://arxiv.org/html/2508.05635v3/Exp-BenchRank-left.png",
     "type": "paper",
     "publisher": "arXiv (AgiBot Genie Team)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "EWMBench Figure 6 (human rank and EWMBench vs VBench scores)",
     "url": "https://arxiv.org/html/2505.09694v2/bar_chart.png",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Qwen-RobotWorld Technical Report (Section 5.1.1 and Table 2: EWMBench)",
     "url": "https://arxiv.org/abs/2606.17030",
     "type": "paper",
     "publisher": "arXiv (Qwen Team)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "AgiBot World Challenge @ ICRA 2026, World Model track test server (evaluation rules)",
     "url": "https://agibot-world-icra26wm.hf.space/",
     "type": "leaderboard",
     "publisher": "AgiBot",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "World Model baseline README for AgiBot World Challenge @ IROS 2025 (commit 62eb4ebfec)",
     "url": "https://github.com/AgibotTech/AgiBotWorldChallengeICRA2026-WorldModelBaseline/blob/62eb4ebfec/README.md",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-07-15",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "WorldArena: A Unified Benchmark for Evaluating Perception and Functional Utility of Embodied World Models",
     "url": "https://arxiv.org/abs/2602.08971",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "WorldArena 2.0: Extending Embodied World Model Benchmarking on Modality, Functionality and Platform (Section 2.2)",
     "url": "https://arxiv.org/html/2605.17912v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Wow, wo, val! A Comprehensive Embodied World Model Evaluation Turing Test",
     "url": "https://arxiv.org/abs/2601.04137",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position",
     "url": "https://arxiv.org/abs/2606.15032",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Pelican-Sim 1.0: A General World Model Simulator for Embodied Intelligence",
     "url": "https://arxiv.org/abs/2609.12036",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "EVEWorld: Physical Evolution Supervision for Embodied World Models",
     "url": "https://arxiv.org/abs/2610.03374",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "ConfAL-WM: Confidence-Guided Active Learning for Action-Conditioned World Models",
     "url": "https://arxiv.org/abs/2608.25572",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "AgiBot World Colosseo paper (robots used for data collection)",
     "url": "https://arxiv.org/html/2503.06669",
     "type": "paper",
     "publisher": "arXiv (AgiBot World team)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "AgiBot World README (licence section)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/blob/main/README.md",
     "type": "repo",
     "publisher": "OpenDriveLab / AgiBot",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Genie Envisioner arXiv abstract page (authors, v1 2025-08-07 to v3 2025-11-04)",
     "url": "https://arxiv.org/abs/2508.05635",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "EWMBench full text, arXiv v1 (for version comparison)",
     "url": "https://arxiv.org/html/2505.09694v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "EWMBench GitHub issues list",
     "url": "https://github.com/AgibotTech/EWMBench/issues?q=is%3Aissue",
     "type": "repo",
     "publisher": "AgiBot",
     "date": "2025-11",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/world-models.json. Changes from the basic entry: status set to dormant (no update for over a year); the '4,644 rows' are frames, so test-set size now separates the paper's 100 episodes from the public 21; embodiment adds humanoid and scene adds kitchen; alias WMBM added. New findings: identical scores credited to EnerVerse_FT and GE-Base, copied Cosmos row in Qwen-RobotWorld, code-versus-paper differences, figure-versus-table mismatch in the human study, and the AGPL-3.0 upstream of the detector."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "gemini-robotics-evals",
   "name": "Gemini Robotics evals",
   "full_name": "Gemini Robotics evaluations (1.0, 1.5, 2)",
   "aliases": [
    "Gemini Robotics generalization benchmark",
    "Gemini Robotics 1.5 benchmark",
    "GR 1.5 evaluation"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Produces scores for embodied policies on real robots. It is a closed, self-run study, so it belongs under 'one-off study', not 'benchmark'.",
   "summary": {
    "text": "Google DeepMind's in-house real-robot tests of its Gemini Robotics models; task suites are described but not released.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Google DeepMind (Gemini Robotics Team)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.5 author line: 'Gemini Robotics Team' with 171 other contributors."
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "No single headquarters stated on pages checked; offices on three continents."
    },
    "first_release": {
     "value": "2025-03 (Gemini Robotics tech report arXiv 2503.20020 v1 2025-03-25; blog 2025-03-12)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Blog date from https://deepmind.google/blog/gemini-robotics-brings-ai-into-the-physical-world/."
    },
    "latest_update": {
     "value": "2026-07: Gemini Robotics 2 blog (2026-07-30) publishes new success rates per skill category; Gemini Robotics 2 safety report dated 2026-07-29",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "No GR 2 VLA technical paper found. Last full methodology: GR 1.5 report, arXiv 2510.03342 v3 (2025-11-28)."
    },
    "version": {
     "value": "Gemini Robotics 2 evaluation (blog charts); methodology last documented in Gemini Robotics 1.5 report v3",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Versions follow model releases: GR 1.0 (2025-03), GR 1.5 (2025-10, v3 2025-11), GR 2 (2026-07)."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.5 Section 2.3: 'we perform A/B/n testing on real robots' for all comparisons in the report. Over 90% of development-time evaluation episodes ran in MuJoCo simulation, but reported comparisons are real."
    },
    "capability": {
     "value": [
      "manipulation",
      "dexterous",
      "bimanual",
      "instruction-following",
      "long-horizon",
      "locomotion"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.5 axes: visual, instruction, action and task generalization; cross-embodiment; multi-step. GR 2 blog adds whole-body humanoid pick-up from floor/shelf and multi-finger dexterity (https://deepmind.google/blog/gemini-robotics-2-brings-whole-body-intelligence-to-robots/)."
    },
    "embodiment": {
     "value": [
      "bimanual-arm",
      "humanoid",
      "dexterous-hand",
      "cross-embodiment"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.5 embodiments: ALOHA, Bi-arm Franka, Apollo humanoid (https://arxiv.org/html/2510.03342). GR 2: Apollo 2 with SharpaWave hands, Apollo 2 with Inspire hands, Franka Duo. On-Device 2 adaptation: SO101, Dexmate (https://deepmind.google/models/model-cards/gemini-robotics-on-device-2/)."
    },
    "scene": {
     "value": [
      "tabletop",
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Tasks span tabletop manipulation and settings such as a laundry room (GR 1.0); long-horizon tasks include trash sorting, desk organization, suitcase packing (GR 1.5); GR 2 shows a garage and a cluttered room."
    },
    "scale": {
     "value": "GR 1.0: generalization benchmark 85 tasks; dexterity benchmark 20 tasks; 20 trials per task for specialist long-horizon tasks (12 for the spelling game). GR 1.5: full benchmark 230 tasks across all embodiments; ALOHA generalization = 68 prior tasks + 5 new action tasks + 12 new task-generalization tasks; long-horizon benchmarks of 4 tasks on ALOHA and 4 on Bi-arm Franka.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.0 numbers from https://arxiv.org/html/2503.20020. GR 2 blog gives no task or trial counts."
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GR 1.0: models are run back-to-back in random order on the same cell (A/B testing) and compared with a pairwise t-test (https://arxiv.org/html/2503.20020). Long-horizon progress is the sum of subtask points. Organiser-run, self-reported. From chart alt text. Each bar averages several tasks in a skill category (multi-finger bars are single tasks). Error bars shown without values. Trial counts and sim/real not stated on the blog or the VLA page (https://deepmind.google/models/gemini-robotics/vla/). Composite index; VQA answers graded by Gemini 2.5 Flash. See the separate ERQA record."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Results only in reports and blogs."
    },
    "access": {
     "value": "closed",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Task suites, rubrics are described in appendices, but robots, scenes and evaluation code are not released."
    },
    "license_code": {
     "value": "not released",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "No evaluation code release found in the reports or blogs."
    },
    "license_data": {
     "value": "not released",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Reported scores are real-robot A/B/n tests. Internally, >90% of development episodes ran in MuJoCo; GR 1.5 Fig. 21 shows rank consistency between sim and real A/B pairs, without a statistic (self-measured; simulator not released)."
    },
    "real_reproducibility": {
     "value": "none",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Single organisation, unreleased setups; outsiders cannot reproduce."
    },
    "kind": {
     "value": "study",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Gemini Robotics 1.5: Pushing the Frontier of Generalist Robots with Advanced Embodied Reasoning, Thinking, and Motion Transfer (full text)",
     "url": "https://arxiv.org/html/2510.03342",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s2": {
     "title": "Gemini Robotics 1.5: Pushing the Frontier of Generalist Robots with Advanced Embodied Reasoning, Thinking, and Motion Transfer",
     "url": "https://arxiv.org/abs/2510.03342",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s3": {
     "title": "Careers at Google DeepMind — Google DeepMind",
     "url": "https://deepmind.google/careers/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Gemini Robotics: Bringing AI into the Physical World",
     "url": "https://arxiv.org/abs/2503.20020",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-03"
    },
    "s5": {
     "title": "Gemini Robotics 2 brings whole body intelligence to robots — Google DeepMind",
     "url": "https://deepmind.google/blog/gemini-robotics-2-brings-whole-body-intelligence-to-robots/",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "Gemini Robotics: Bringing AI into the Physical World (full text)",
     "url": "https://arxiv.org/html/2503.20020",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-03"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/search?query=Gemini%20Robotics%20Bringing%20AI%20into%20the%20Physical%20World",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "genie-sim",
   "name": "Genie Sim",
   "full_name": "Genie Sim Benchmark (AgiBot)",
   "aliases": [
    "Genie Sim 3.0",
    "GenieSim Benchmark",
    "RoboColiseum",
    "GenieSim 2.2 (AgiBot World Challenge 2025 sim track)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores manipulation policies in closed-loop simulation and runs a public leaderboard (RoboColiseum).",
   "summary": {
    "text": "AgiBot's simulation benchmark scoring robot policies on instruction, spatial, robustness and manipulation boards, with a VLM judge.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "AgiBot (README: 'Genie Sim is the simulation platform from AgiBot'); 19 authors listed on arXiv, affiliations not shown on the abs page",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "builder_type": {
     "value": "robot-company",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Data hosted on ModelScope; support via Feishu/WeChat groups. Company HQ not checked at a primary source today. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2025-04: repo created 2025-04-21 (earliest commits April 2025). Release log starts at v2.1 2025-06-25 (10 more manipulation tasks for AgiBot World Challenge 2025) and v2.2 2025-07-14 (evaluation metrics for all challenge tasks). Genie Sim 3.0: arXiv v1 2026-01-05, CES announcement 2026-01-06.",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "First public date taken from repo creation (GitHub API); no v1.0 entry in the release log. Press release: https://agibot.com/article/231/detail/29.html Checked 2026-10-10."
    },
    "latest_update": {
     "value": "v3.2.0: release-log entry dated 2026-06-25 opens RoboColiseum and adds the 'spatial' board; GitHub release tag v3.2.0 published 2026-08-04; arXiv v4 2026-08-14; last push 2026-09-07.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "version": {
     "value": "v3.2.0 (VERSION file)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "arXiv preprint only (no comments or journal-ref on abs page)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Runs in NVIDIA Isaac Sim; README also lists Newton backends. Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following",
      "embodied-reasoning",
      "bimanual"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Boards: Instruction (colour/size/shape/number/logic/common-sense picks), Spatial (relative position, sorting, stacking), Robust, Manipulation (incl. 'Bimanual Chip Handover' in the Sim2Real table). Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "humanoid",
      "bimanual-arm"
     ],
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Paper: data generated on AgiBot G1 and G2; title says 'for Humanoid Robot'. README: Genie G2 family is Tier 1; reference URDFs for Franka, UR5, Aloha, ARX, Agilex. Form factor of G1/G2 not checked at a primary source today. Checked 2026-10-10."
    },
    "robots": {
     "value": "AgiBot G1, AgiBot G2 (paper); reference URDFs for Franka, UR5, Aloha, ARX, Agilex (README)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "retail-logistics",
      "industrial",
      "home",
      "office-lab",
      "kitchen"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "README: assets cover retail, industry, catering, home and office. 'catering' mapped to kitchen. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Board scores are 0-1 per task. A VLM judges whether task requirements are met. v2.2 release log: evaluation script records the score of all steps. Exact aggregation formula not given. Checked 2026-10-10."
    },
    "trials": {
     "value": "50 trials per configuration in the Sim2Real study; board trial counts not stated",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "evaluator": {
     "value": null,
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "organiser-run on RoboColiseum: users submit a model and the pipeline runs automatically; README: debug locally, then launch an official evaluation against your inference server",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Official (: RoboColiseum (overall + four sub-boards). Leaderboard table showed no rows when loaded 2026-10-10. README lists baseline scores, e.g. Instruction avg ACoT-VLA 0.76, pi0.5 0.75, GR00T-N1.7 0.65, pi0 0.37; Manipulation avg pi0.5 0.58.)",
     "note": "Baseline numbers are publisher-run. Checked 2026-10-10."
    },
    "license_code": {
     "value": "MPL-2.0 for source/geniesim* and source/data_collection; source/scene_reconstruction has multiple licences",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. GitHub API reports NOASSERTION. Checked 2026-10-10."
    },
    "license_data": {
     "value": "CC BY-NC-SA 4.0 stated in the ModelScope dataset card README and in the gated Hugging Face asset card (AgiBot World Community License prompt). ModelScope metadata field says 'Apache License 2.0'.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Conflict between card text and metadata field; card text recorded as primary. Asset card: https://huggingface.co/datasets/agibot-world/GenieSimAssets Checked 2026-10-10."
    },
    "access": {
     "value": "registration",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Registration (Hugging Face asset repo is gated with a form); ModelScope dataset public)",
     "note": "Checked 2026-10-10."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "From CC BY-NC-SA 4.0 on data/assets; code MPL-2.0 allows commercial use. Not legal advice. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "correlated",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Measured (: R^2 = 0.931, slope ~1.023 over 16 model configurations (pi0.5 fine-tuned on 4 tasks with 200/500 real or 500/1500 sim episodes), 50 trials each in sim and real; measured by the authors. README Sim2Real table: 8 tasks, 2x2 data-source x environment design, averages sim-to-real 0.83 vs real-to-real 0.75.)",
     "note": "Paper Fig. 7: 16 model configurations, R^2 = 0.931, slope ~1.023 between sim and real scores; 50 trials per configuration in each. All 16 are pi0.5 fine-tuned on 4 tasks with 4 data mixes, measured by the AgiBot authors. It tests one policy family, not a ranking of different policies. RoboColiseum site claims a sim-to-real gap under 10% without showing a statistic."
    },
    "citations": {
     "value": 15,
     "display": "15",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 1416,
     "display": "1416",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "AgibotTech/genie_sim on GitHub (repository)",
     "url": "https://github.com/AgibotTech/genie_sim",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s2": {
     "title": "GenieSim3.0-Dataset - ModelScope 魔搭社区",
     "url": "https://modelscope.cn/datasets/agibot_world/GenieSim3.0-Dataset",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Genie Sim 3.0 : A High-Fidelity Comprehensive Simulation Platform for Humanoid Robot",
     "url": "https://arxiv.org/abs/2601.02078",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s4": {
     "title": "Genie Sim 3.0 : A High-Fidelity Comprehensive Simulation Platform for Humanoid Robot (full text)",
     "url": "https://arxiv.org/html/2601.02078v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s5": {
     "title": "RoboColiseum",
     "url": "https://robocoliseum.ai/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "AgibotTech/genie_sim on GitHub (blob)",
     "url": "https://github.com/AgibotTech/genie_sim/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "agibot-world/GenieSimAssets on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/agibot-world/GenieSimAssets",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2601.02078",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "genmanip",
   "name": "GenManip",
   "aliases": [
    "GenManip-Bench",
    "GenManip Suite",
    "GENMANIP"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "GenManip-Bench scores manipulation policies (modular and end-to-end) on fixed simulated scenarios with SR/SPL; the platform also hosts further scored benchmarks.",
   "summary": {
    "text": "Isaac Sim tabletop platform that generates instruction-following manipulation tasks with an LLM; its benchmark has 200 scenarios.",
    "sources": [
     "s6"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Shanghai AI Laboratory (9 of 10 authors); co-affiliations Zhejiang University (Haifeng Huang), Xi'an Jiaotong University and Nanjing University (Jiangmiao Pang)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Shanghai AI Laboratory is a public research lab."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-06 (arXiv v1 2025-06-12; CVPR 2025 proceedings, June 2025)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Repo was created 2025-04-22 (GitHub API)."
    },
    "latest_update": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Main-branch commit 2026-09-15: 'update ebench task bottle and shop max num_steps to 5000'; .version reads 0.2.3-260320-alpha",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "Archived CVPR README: Oct 2025 data-synthesis pipeline and evaluation toolkit released; Aug 2025 IROS 2025 challenge integration",
       "level": "verified",
       "sources": [
        "s4"
       ]
      }
     ]
    },
    "version": {
     "value": "0.2.3-260320-alpha (main branch .version file); no GitHub releases",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "CVPR 2025, pp. 12187-12198",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "arXiv abs page lists no venue."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Paper embodiment from arXiv HTML; IROS from archived README."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Abstract: 'a realistic tabletop simulation platform'."
    },
    "scoring": {
     "value": [
      "progress",
      "path-efficiency",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Trials per scenario not stated."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (for GenManip-Bench; no leaderboard found on genmanip.com; a time-bound IROS 2025 challenge leaderboard ran on eval.ai)",
     "note": "Archived README ticks 'Website, documentation, and leaderboard' but the docs site shows no leaderboard; successor EBench links an online leaderboard at internrobotics.shlab.org.cn/eval."
    },
    "license_code": {
     "value": "MIT on branch archived/cvpr2025 (copyright line is the template placeholder '[Your Name]'); main branch has no LICENSE file and GitHub reports no licence",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "CC-BY-NC-SA-4.0 (InternRobotics/IROS-2025-Challenge-Manip dataset card)",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "unknown for GenManip Suite packages and datasets (GenManipSuite/*, Axi404/GenManip-Dataset-*): no licence tag on Hugging Face",
       "level": "unknown",
       "sources": [],
       "note": "Checked HF API metadata for 3 package repos and 2 dataset repos."
      }
     ]
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s12"
     ],
     "checked": "2026-10-10",
     "display": "non-commercial for challenge data (CC-BY-NC-SA); unclear for main-branch code",
     "note": "Not legal advice."
    },
    "sim_to_real": {
     "value": "claimed",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Claimed (: one real-world Franka deployment of the modular baseline shown as a figure; sim-vs-real comparison left to future work)",
     "note": "Supplementary Sec. 12 shows one real-world deployment of the prompt-based modular baseline on a Franka ('chosen result' figure, no numbers in text). Authors leave sim-vs-real comparison to future work and describe a notable sim-to-real gap (physics parameters need tuning). No paired measurements; no follow-up found."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s15"
     ],
     "checked": "2026-10-10",
     "display": "Active (as a platform)",
     "note": "Commits in 2026-09; original GenManip-Bench itself appears superseded by newer hosted benchmarks."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "GENMANIP: LLM-driven Simulation for Generalizable Instruction-Following Manipulation (full text)",
     "url": "https://arxiv.org/html/2506.10966v1",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s2": {
     "title": "GENMANIP: LLM-driven Simulation for Generalizable Instruction-Following Manipulation",
     "url": "https://arxiv.org/abs/2506.10966",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s3": {
     "title": "InternRobotics/GenManip on GitHub (commits?per_page=3)",
     "url": "https://api.github.com/repos/InternRobotics/GenManip/commits?per_page=3",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "InternRobotics/GenManip on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/InternRobotics/GenManip/archived/cvpr2025/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "InternRobotics/GenManip on GitHub (file .version)",
     "url": "https://raw.githubusercontent.com/InternRobotics/GenManip/main/.version",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "CVPR 2025 Open Access Repository",
     "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Gao_GENMANIP_LLM-driven_Simulation_for_Generalizable_Instruction-Following_Manipulation_CVPR_2025_paper.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "InternRobotics/GenManip on GitHub (repository)",
     "url": "https://github.com/InternRobotics/GenManip",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "CVPR Poster GENMANIP: LLM-driven Simulation for Generalizable Instruction-Following Manipulation",
     "url": "https://cvpr.thecvf.com/virtual/2025/poster/33632",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Benchmarks",
     "url": "https://genmanip.com/evaluation/benchmarks/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "https://eval.ai/api/challenges/challenge/2626/",
     "url": "https://eval.ai/api/challenges/challenge/2626/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "InternRobotics/GenManip on GitHub (archived)",
     "url": "https://raw.githubusercontent.com/InternRobotics/GenManip/archived/cvpr2025/LICENSE",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s12": {
     "title": "InternRobotics/IROS-2025-Challenge-Manip on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/InternRobotics/IROS-2025-Challenge-Manip",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s13": {
     "title": "InternRobotics/EBench on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/InternRobotics/EBench/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s14": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2506.10966",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s15": {
     "title": "InternRobotics/GenManip on GitHub (repository)",
     "url": "https://api.github.com/repos/InternRobotics/GenManip",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "grutopia",
   "name": "GRUtopia",
   "aliases": [
    "InternUtopia",
    "GRBench",
    "GRScenes",
    "GRResidents"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Its benchmark GRBench scores embodied agents (legged robots) on navigation, social navigation and mobile manipulation in simulation.",
   "summary": {
    "text": "Isaac Sim platform of large interactive scenes with LLM-driven characters; its benchmark tests legged robots navigating and fetching objects.",
    "sources": [
     "s2"
    ]
   },
   "facts": {
    "publishers": {
     "value": "22 authors, all with OpenRobotLab, Shanghai AI Laboratory; co-affiliations include Nanjing University, The Chinese University of Hong Kong, Xidian University, Shanghai Jiao Tong University and Tsinghua University",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Shanghai AI Laboratory is a public research lab."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-07 (arXiv v1 2024-07-15; README news: paper and demos released 2024-07)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "v2.2.1 (2025-09-04) bug fixes; v2.2.0 (2025-07-25): multi-episode, vector env, multi-GPU, Isaac Sim 4.5.0, packages renamed grutopia.* -> internutopia.*",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "InternUtopia v2.2.1; GitHub URL OpenRobotLab/GRUtopia now resolves to InternRobotics/InternUtopia",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "arXiv preprint only (README BibTeX booktitle 'arXiv'); no venue on arXiv page",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar also lists arXiv.org."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "navigation",
      "mobile-manipulation",
      "collaboration",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid",
      "legged",
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home",
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Sec. 4.1 states success as 'target in field of view' without the 3 m rule; Sec. 4.3 adds it."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: README points to InternNav and InternManip to run the benchmarks; no GRBench leaderboard found)",
     "note": "InternNav user guide does not mention GRBench (checked 2026-10-10)."
    },
    "license_code": {
     "value": "MIT ('Copyright (c) Intern Robotics, Shanghai AI Laboratory')",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "CC-BY-NC-SA-4.0 (GRScenes dataset card); README requires a user agreement form before GRScenes-100 access",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "allowed for code; non-commercial for scenes",
     "note": "Not legal advice."
    },
    "sim_to_real": {
     "value": "claimed",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Claimed (: Appendix E real-world demo on Unitree H1 (object loco-navigation), no numbers)",
     "note": "Appendix E: the LLM-agent baseline was demonstrated on a real Unitree H1 for object loco-navigation, using the same locomotion policy as in simulation; described as 'smoothly transferred', no numbers, video promised later. No paired sim/real measurements found."
    },
    "status": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "display": "superseded as a named benchmark; platform renamed InternUtopia and dormant since 2025-09",
     "note": "From release dates and README pointers to InternNav/InternManip; InternNav guide names only VL-LN Bench as an extended benchmark."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "GRUtopia: Dream General Robots in a City at Scale (full text)",
     "url": "https://arxiv.org/html/2407.10943v1",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-07"
    },
    "s2": {
     "title": "GRUtopia: Dream General Robots in a City at Scale",
     "url": "https://arxiv.org/abs/2407.10943",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-07"
    },
    "s3": {
     "title": "InternRobotics/InternUtopia on GitHub (releases)",
     "url": "https://api.github.com/repos/InternRobotics/InternUtopia/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "InternRobotics/InternUtopia on GitHub (repository)",
     "url": "https://api.github.com/repos/InternRobotics/InternUtopia",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "InternRobotics/InternUtopia on GitHub (readme)",
     "url": "https://api.github.com/repos/InternRobotics/InternUtopia/readme",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "InternRobotics/GRScenes on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/InternRobotics/GRScenes",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s7": {
     "title": "InternRobotics/InternUtopia on GitHub (license)",
     "url": "https://api.github.com/repos/InternRobotics/InternUtopia/license",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "InternNav — Intern Robotics Documentation v0.0.1 documentation",
     "url": "https://internrobotics.github.io/user_guide/internnav/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2407.10943",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "habitat-3-0",
   "name": "Habitat 3.0",
   "full_name": "Habitat 3.0: A Co-Habitat for Humans, Avatars and Robots",
   "aliases": [
    "Habitat3",
    "Hab3",
    "Social Navigation (Habitat 3.0)",
    "Social Rearrangement (Habitat 3.0)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "Habitat 3.0 is Meta FAIR's 2023 simulator release that adds simulated people to Habitat. It defines two tasks for a simulated Spot robot: finding and following a person (Social Navigation) and tidying a home together with a person (Social Rearrangement). Real people can join through a human-in-the-loop tool.",
    "sources": [
     "s1",
     "s2",
     "s4"
    ],
    "short": "Habitat 3.0 is a simulator release with two tasks in which a simulated Spot robot works with simulated people. The robot either finds and follows a person or tidies a home together with one."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s4",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The project page presents the two tasks 'aiming at reproducible and standardized benchmarking'. The paper calls Habitat 3.0 a simulation platform."
    },
    "kind_secondary": {
     "value": [
      "platform"
     ],
     "display": "Also a simulator release (habitat-sim and habitat-lab v0.3) with humanoid avatars and a human-in-the-loop tool.",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The platform is shared with other Habitat tasks, so its maintenance status applies to them too."
    },
    "publishers": {
     "value": [
      "Meta FAIR"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "23 authors. The paper's footnote says 'Work done at Fair, Meta'; no other affiliations are given. The launch blog is a Meta AI post.",
     "short": "Meta FAIR"
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Meta FAIR is an industry AI research lab."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "Meta Platforms' principal executive offices are in Menlo Park, California (FY2025 10-K cover page). The paper does not say where the team sits."
    },
    "first_release": {
     "value": "2023-10",
     "display": "arXiv v1 on 2023-10-19. habitat-sim v0.3.0 (2023-10-19) and habitat-lab v0.3.0 (2023-10-20) shipped it; the Meta blog post followed on 2023-10-20.",
     "level": "verified",
     "sources": [
      "s1",
      "s6",
      "s5",
      "s35"
     ],
     "checked": "2026-10-10",
     "short": "October 2023, at ICLR 2024"
    },
    "published_at": {
     "value": "ICLR 2024",
     "display": "ICLR 2024 (poster). arXiv has only v1.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "short": "ICLR 2024"
    },
    "latest_update": {
     "value": "2026-05",
     "display": "habitat-lab v0.3.4 on 2026-05-07. The same day a README notice said Meta no longer develops or maintains the project beyond v0.3.4. The task episodes last changed on 2023-11-28.",
     "level": "verified",
     "sources": [
      "s6",
      "s9",
      "s12",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "The habitat-sim README carries the same notice. habitat-sim's last release is v0.3.3 (2026-02-12).",
     "short": "May 2026. This was Meta's final release."
    },
    "version": {
     "value": "habitat-lab v0.3.4; habitat-sim v0.3.3",
     "level": "verified",
     "sources": [
      "s6",
      "s35",
      "s14"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v0.3.0",
       "display": "2023-10-19 (sim) and 2023-10-20 (lab): first release with Habitat 3.0.",
       "level": "verified",
       "sources": [
        "s6",
        "s35"
       ]
      },
      {
       "value": "v0.3.1 to v0.3.3",
       "display": "habitat-lab 2024-03-15, 2024-10-30, 2025-01-27; habitat-sim 2024-03-15, 2024-10-30, 2026-02-12.",
       "level": "verified",
       "sources": [
        "s6",
        "s35"
       ]
      },
      {
       "value": "habitat-lab v0.3.4",
       "display": "2026-05-07. Last release maintained by Meta.",
       "level": "verified",
       "sources": [
        "s6",
        "s9"
       ]
      },
      {
       "value": "hab3_episodes",
       "display": "Episode data for both tasks plus a social navigation checkpoint. Last modified 2023-11-28.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      }
     ],
     "short": "habitat-lab v0.3.4 and habitat-sim v0.3.3"
    },
    "capability": {
     "value": [
      "collaboration",
      "navigation",
      "mobile-manipulation"
     ],
     "level": "verified",
     "sources": [
      "s4",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Social Navigation: find and follow a person while keeping 1 to 2 m away. Social Rearrangement: move objects together with a person."
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "object-pose"
     ],
     "display": "Evaluation uses homes and object placements not seen in training. Social Rearrangement also tests partners not seen in training (zero-shot coordination), which has no taxonomy value.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "Homes and partners not seen in training"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All experiments, including the human-in-the-loop study, run in simulation."
    },
    "simulator": {
     "value": "Habitat 3.0 (habitat-sim and habitat-lab v0.3)",
     "display": "Habitat-Sim and Habitat-Lab v0.3. Simulated people use SMPL-X bodies with walking and reaching motions. The paper reports 1190 FPS with a humanoid and a robot (1345 FPS with two robots).",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The intro gives 1190 FPS. Appendix F.1 reports 1,100 to 2,290 FPS with 16 environments on one V100 GPU.",
     "short": "Habitat-Sim and Habitat-Lab v0.3"
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "display": "Simulated Boston Dynamics Spot with an arm. Its legs are not simulated; the base moves by velocity commands. In Social Navigation only the base moves. Partners are simulated people, or real people through the human-in-the-loop tool.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The basic entry is kept. The partner is a simulated person, not a robot, so 'humanoid' is not added."
    },
    "robots": {
     "value": "Boston Dynamics Spot with arm (simulated)",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "The Spot model (URDF) is 'provided courtesy of Boston Dynamics, all rights reserved', redistributed with written permission (dataset card).",
     "short": "Spot (simulated)"
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 2,
     "display": "2 tasks: Social Navigation (find a person and follow at 1 to 2 m) and Social Rearrangement (move two objects to goals together with a person).",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "2 tasks"
    },
    "scenes": {
     "value": 59,
     "display": "59 HSSD homes: 37 train, 12 validation and 10 test, taken from the 211 scenes of HSSD-200",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "59 homes: 37 for training, 12 for validation and 10 for testing"
    },
    "objects": {
     "value": "YCB objects",
     "display": "Objects are limited to the YCB set 'for simplicity' (paper Appendix E).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "YCB objects"
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Following rate, collision rate and relative efficiency have no taxonomy value."
    },
    "metric_detail": {
     "value": "Several metrics per task",
     "display": "Social Navigation: finding success S, SPS (S weighted by path steps against an oracle that knows the person's route), following rate F (share of possible steps spent 1 to 2 m away facing the person) and collision rate CR (share of episodes ending in a collision). Social Rearrangement: success rate SR (both objects placed) and relative efficiency RE (steps for the person alone compared with the team), tested with the training partners and with 10 partly unseen partners.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "SDA (2024) adds episode success ES: find the person and follow for 400 steps at a safe distance.",
     "short": "Rates of finding, following and collision, and success at the joint task"
    },
    "trials": {
     "value": "1,500-step episodes; 3 training seeds",
     "level": "verified",
     "sources": [
      "s2",
      "s10",
      "s25"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Social Navigation episodes",
       "display": "1,500 steps, ending early on a collision. The robot starts at least 5 m from the person. Start positions and the person's path are fixed across baselines.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Seeds",
       "display": "Every baseline is trained with 3 random seeds and results are averaged.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Episode counts differ by source",
       "display": "Paper appendix: 15 episodes in each of 10 test scenes (150) and 100 in each of 12 validation scenes (1,200). habitat-baselines README: the paper used 'the full evaluation dataset (1200 episodes)' and its own command evaluates 500. CommNav (2026): 400 test episodes over 12 environments.",
       "level": "verified",
       "sources": [
        "s2",
        "s10",
        "s14",
        "s25"
       ],
       "note": "See issues.i1."
      }
     ],
     "short": "1,500-step episodes, with 3 training seeds"
    },
    "uncertainty_reported": {
     "value": "yes",
     "level": "inferred",
     "sources": [
      "s2",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "The paper reports means with ± over 3 seeds and 95% confidence intervals for the human study. SDA reports ± too. We checked these two papers only."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s4",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions. Each paper runs its own evaluation."
    },
    "leaderboard": {
     "value": "paper-only",
     "display": "No leaderboard or challenge for these two tasks. Results live in papers.",
     "level": "inferred",
     "sources": [
      "s4",
      "s31",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Checked the project page, the habitat-lab README and aihabitat.org/challenge, which redirects to the 2023 HomeRobot OVMM challenge.",
     "short": "Results appear only in papers."
    },
    "top_score": {
     "value": "No single headline score",
     "display": "No single headline score. Social Navigation finding success: 0.97 for the paper's RL baseline with the person's position as input, 0.91 for SDA without it. Social Rearrangement: 71.79% success with unseen partners (Plan-Pop3).",
     "level": "verified",
     "sources": [
      "s2",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Values are as printed. Metrics, inputs and partners differ by row, so they are not one ranking.",
     "items": [
      {
       "value": "Heuristic expert (privileged map)",
       "display": "S 1.00, SPS 0.97, F 0.51, CR 0.52",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "End-to-end RL with humanoid GPS",
       "display": "S 0.97±0.00, SPS 0.65±0.00, F 0.44±0.01, CR 0.51±0.03 (2023)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "End-to-end RL without humanoid GPS",
       "display": "S 0.76±0.02, SPS 0.34±0.01, F 0.29±0.01, CR 0.48±0.03 (2023)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "SDA-S2, no privileged inputs",
       "display": "S 0.91±0.01, SPS 0.45±0.01, F 0.39±0.01, CR 0.57±0.02, ES 0.43±0.02 (2024; ICLR 2025)",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "Social Rearrangement, Plan-Pop3",
       "display": "Unseen partners: SR 71.79±7.38%, RE 101.99±15.18. Training partners: SR 77.79±2.86%.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Social Rearrangement, Learn-Single",
       "display": "Training partner: SR 98.50±0.48%, RE 159.2±1.0. Unseen partners: SR 50.94±39.55%.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "There is no single score. The best learned policy reaches a finding success of 0.97. A rule-based expert with a privileged map reaches 1.00."
    },
    "human_in_the_loop": {
     "value": "30 participants",
     "display": "30 people each did 10 episodes per condition by keyboard and mouse: alone, with a Learn-Single robot, and with a Plan-Pop3 robot. Both robots made people faster (RE 133.80 and 123.46). The order of the two policies on RE and on the share of work done by the robot matched the automated test.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "This checks the simulated partner against real people, with simulated robots. It is not a real-robot test.",
     "items": [
      {
       "value": "Real people adapt",
       "display": "Real partners reached success 1 in all episodes. The authors conclude their simulated partners 'do not accurately capture' human-robot dynamics.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Significance",
       "display": "Task steps differed significantly between working alone and either robot; the difference between the two robots was not significant (p = 0.0533).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "30 people took part. The two policies ranked in the same order as in the automated test."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s11",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "habitat-lab LICENSE: MIT, 'Copyright (c) Meta Platforms, Inc. and its affiliates'. habitat-sim is MIT per the GitHub API.",
     "short": "MIT"
    },
    "license_data": {
     "value": [
      "CC-BY-NC-4.0"
     ],
     "display": "Task episodes (hab3_episodes) and benchmark assets (hab3_bench_assets): CC BY-NC 4.0.",
     "level": "verified",
     "sources": [
      "s14",
      "s15"
     ],
     "checked": "2026-10-10",
     "short": "CC BY-NC 4.0"
    },
    "license_assets": {
     "value": "CC-BY-NC-4.0 and others",
     "display": "Scenes (HSSD) CC BY-NC 4.0. Humanoid avatars non-commercial, with conflicting labels. Spot model by permission of Boston Dynamics. YCB objects CC BY 4.0.",
     "level": "verified",
     "sources": [
      "s13",
      "s16",
      "s17",
      "s18"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "HSSD scenes",
       "display": "CC BY-NC 4.0. The card asks users to acknowledge the licence.",
       "level": "verified",
       "sources": [
        "s13"
       ]
      },
      {
       "value": "Humanoid avatars: conflict",
       "display": "Card metadata says CC BY-NC-SA 4.0; the card text says the 12 avatars and reaching poses are CC BY-NC 4.0, and the walking motion is under the SMPL Body Motion File License.",
       "level": "verified",
       "sources": [
        "s16"
       ],
       "note": "CONFLICT within one card."
      },
      {
       "value": "Spot model",
       "display": "Licence 'other': Boston Dynamics, all rights reserved; written permission for redistribution with attribution.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "YCB objects",
       "display": "CC BY 4.0",
       "level": "verified",
       "sources": [
        "s18"
       ]
      }
     ],
     "short": "Non-commercial for the HSSD scenes and the avatars"
    },
    "access": {
     "value": "open",
     "display": "Code on GitHub; episodes, avatars and scenes on Hugging Face, downloadable without an account through habitat-sim's download script.",
     "level": "verified",
     "sources": [
      "s14",
      "s19",
      "s21",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "The HSSD card shows a licence acknowledgement prompt; the Hub API reports the dataset as not gated (2026-10-10).",
     "short": "Open. It can be downloaded without an account."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s13",
      "s14",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Scenes, episodes and avatars are licensed for non-commercial use only. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No real-robot comparison found. The paper's human-in-the-loop study used real people with a simulated robot.",
     "level": "inferred",
     "sources": [
      "s2",
      "s5",
      "s23",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "The paper has no real-robot experiments; Meta's blog lists deploying the learned models in the physical world as a next step. The blog's '20% success rate in the physical world' refers to the HomeRobot OVMM benchmark, not these tasks.",
     "short": "We found no comparison with real robots."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "citations": {
     "value": 342,
     "display": "342 (Semantic Scholar; 87 influential)",
     "level": "verified",
     "sources": [
      "s22"
     ],
     "checked": "2026-10-10",
     "short": "342"
    },
    "github_stars": {
     "value": 3154,
     "display": "3,154 stars, 690 forks (habitat-lab); habitat-sim has 3,834 stars",
     "level": "verified",
     "sources": [
      "s7",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The repositories host all Habitat tasks, not only Habitat 3.0.",
     "short": "3,154 for habitat-lab"
    },
    "dataset_downloads": {
     "value": 121,
     "display": "hab3_episodes: 121 (Hub 'downloads' field), 1,417 all time. habitat_humanoids: 1,922, 21,328 all time. hssd-hab: 95,420, 519,213 all time (shared with other tasks).",
     "level": "verified",
     "sources": [
      "s19",
      "s20",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API. We did not check the time window behind the 'downloads' field.",
     "short": "121 for the task episodes on Hugging Face"
    },
    "used_by": {
     "value": "Few papers report the original task metrics. In the citation contexts we scanned, most of the 342 citing papers use Habitat 3.0 as a simulator for other tasks.",
     "level": "inferred",
     "sources": [
      "s33",
      "s23",
      "s25",
      "s27"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "SDA (Sapienza University of Rome et al.)",
       "display": "2024-04; ICLR 2025 spotlight. Social Navigation with the paper's metrics plus episode success.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "CommNav (Habitat 3.0c)",
       "display": "2026-07; IROS 2026. Multi-person variant with robot-to-person questions; reports episode success.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "CST-WM",
       "display": "2026-09. Tracking on Habitat 3.0 with its own protocol and metrics.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      }
     ],
     "short": "A few papers report on the tasks. Most citing papers use it as a simulator."
    },
    "derived_benchmarks": {
     "value": [
      "PARTNR",
      "Social-HM3D / Social-MP3D",
      "Habitat 3.0c (CommNav)"
     ],
     "level": "verified",
     "sources": [
      "s26",
      "s24",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list.",
     "items": [
      {
       "value": "PARTNR",
       "display": "Meta FAIR, 2024-10. 100,000 language tasks for a robot and a person, built on Habitat 3.0.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "Social-HM3D / Social-MP3D (Falcon)",
       "display": "2024-09; ICRA 2025. Point-goal navigation among several moving people in scanned homes, using Habitat 3.0 humans.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "Habitat 3.0c (CommNav)",
       "display": "2026-07; IROS 2026. Adds several people and robot-to-person communication.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      }
     ],
     "short": "PARTNR and social navigation variants"
    },
    "status": {
     "value": "dormant",
     "display": "Meta stopped development after habitat-lab v0.3.4 (2026-05-07). The task data has not changed since 2023-11.",
     "level": "inferred",
     "sources": [
      "s9",
      "s32",
      "s14",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "By the taxonomy's one-year rule the platform was updated recently, but the maintainer has announced no further work, so we use 'dormant'. Forks may continue.",
     "short": "Meta stopped maintaining it in May 2026."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "There is no fixed evaluation set",
     "text": "The paper's appendix lists 150 test episodes (15 in each of 10 test scenes) and 1,200 validation episodes. The habitat-baselines README says the paper's numbers come from 'the full evaluation dataset (1200 episodes)' and gives a command that evaluates 500 episodes. Its released social navigation checkpoint uses Spot's body stereo depth in place of the paper's humanoid GPS, and its expected output (found-human rate 0.9020) is not in the paper's metric format. A 2026 paper evaluates on 400 episodes over 12 environments.",
     "level": "verified",
     "sources": [
      "s2",
      "s10",
      "s25"
     ],
     "status": "open",
     "short": "Sources use 150, 400, 500 or 1,200 evaluation episodes."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "Collision rate penalises following for longer",
     "text": "CR is the share of episodes that end in a collision, and episodes run to 1,500 steps instead of stopping once the robot has followed long enough. The SDA authors note that a robot which finds and follows the person longer has more chances to collide. When they ended episodes after 400 following steps, SDA and the baseline had similar collision rates (0.39 and 0.38). In the original table even the privileged heuristic expert ends 52% of episodes in a collision.",
     "level": "verified",
     "sources": [
      "s23",
      "s2"
     ],
     "status": "open",
     "short": "A robot that follows the person for longer has more chances to collide. So the following rate and the collision rate trade off against each other."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Simulated people behave simply",
     "text": "The simulated people only walk and reach; in Social Navigation the person walks shortest paths to random points. In the human study, real partners adapted to the robot and reached success 1 in all episodes, while simulated partners left success rates of 48.52% to 71.79% with unseen partners. The authors conclude their simulated partners do not accurately capture human-robot dynamics.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "The simulated partners act less like real people than the scores imply."
    },
    {
     "id": "i4",
     "type": "inconsistent-reporting",
     "title": "Text and tables disagree in places",
     "text": "The text gives 71.7% for the best unseen-partner success, the table 71.79. The text says Plan-Pop4's RE drops from 105.49 to 101.99, but 101.99 is Plan-Pop3's value in the table; Plan-Pop4's is 103.53. The text says learned skills drop success to 41.96%, the table shows 41.09. The main text gives the humanoid 188±2 FPS in one environment; Appendix F.1 gives 155±26.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "Several numbers in the text differ from the paper's tables."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "The project is unmaintained and released files have open bugs",
     "text": "Meta stopped maintaining habitat-lab and habitat-sim after v0.3.4 (2026-05-07). Open issues report errors when evaluating the released social navigation checkpoint (issue #1839, 2024-03) or the social navigation baseline (issue #1684, 2023-11), and objects missing from the provided episodes (issue #2105, 2024-11). Users posted workarounds; one says the fix still fails on some scenes.",
     "level": "verified",
     "sources": [
      "s9",
      "s28",
      "s29",
      "s30"
     ],
     "status": "open",
     "short": "Meta has ended maintenance, and reported bugs in the data remain open."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "Habitat 3.0 scores measure how well a policy works with scripted or learned simulated people in simulated homes. The human study suggests the order of two policies carries over to real people in simulation. Nothing links the scores to real robots.",
     "basis": [
      "facts.sim_to_real",
      "facts.human_in_the_loop",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Scores show how well a policy works with simulated people. Nothing links them to real robots."
    },
    {
     "id": "r2",
     "text": "Few papers report on these exact tasks and the evaluation set varies, so published numbers work as reference points. They do not form a leaderboard.",
     "basis": [
      "facts.leaderboard",
      "facts.used_by",
      "issues.i1"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Treat the few published numbers as reference points."
    },
    {
     "id": "r3",
     "text": "Habitat 3.0 matters most as the base for later benchmarks such as PARTNR and Social-HM3D. With Meta's maintenance ended in 2026, new work will depend on forks.",
     "basis": [
      "facts.derived_benchmarks",
      "facts.status"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "It is mostly used as the base for newer benchmarks."
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Habitat 3.0 paper (full text, including Appendix D on the Spot robot and Appendix H), project page, Meta launch blog, SDA (2404.11327), Falcon (2409.13244), CommNav (2607.01044), Semantic Scholar citation contexts for all 342 citing papers filtered for real-robot or hardware mentions, web searches for real Spot deployment of Habitat 3.0 social navigation policies. None found.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project page, habitat-lab and habitat-baselines READMEs, aihabitat.org/challenge (redirects to the 2023 HomeRobot OVMM challenge).",
     "date": "2026-10-10"
    },
    {
     "for": "used_by",
     "where": "Semantic Scholar citation contexts (342 records) scanned for the task metrics (SPS, following rate, relative efficiency, zero-shot coordination); web searches for papers reporting these metrics.",
     "date": "2026-10-10"
    },
    {
     "for": "trials (number of test episodes used for the paper's tables)",
     "where": "Paper Sections 4.1 and 4.2 and Appendices A and E; habitat-baselines README; hab3_episodes card. The figure and table captions do not name the split.",
     "date": "2026-10-10"
    },
    {
     "for": "issues (reproduction problems)",
     "where": "habitat-lab GitHub issues searched for social_nav, social rearrangement and hab3 episodes. OpenReview reviews could not be read (the API returned HTTP 403).",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Habitat 3.0: A Co-Habitat for Humans, Avatars and Robots (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2310.13724",
     "type": "paper",
     "publisher": "arXiv (Meta FAIR)",
     "date": "2023-10-19",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Habitat 3.0 paper, full text v1 (Sections 3 to 5, Appendices A, D, E, F, G, H)",
     "url": "https://arxiv.org/html/2310.13724v1",
     "type": "paper",
     "publisher": "arXiv (Meta FAIR)",
     "date": "2023-10-19",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "ICLR 2024 poster page: Habitat 3.0: A Co-Habitat for Humans, Avatars, and Robots",
     "url": "https://iclr.cc/virtual/2024/poster/19442",
     "type": "paper",
     "publisher": "ICLR 2024",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Habitat 3.0 project page",
     "url": "https://aihabitat.org/habitat3/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Introducing Habitat 3.0: The next milestone on the path to socially intelligent robots",
     "url": "https://ai.meta.com/blog/habitat-3-socially-intelligent-robots-siro/",
     "type": "blog",
     "publisher": "Meta AI",
     "date": "2023-10-20",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "facebookresearch/habitat-lab releases",
     "url": "https://github.com/facebookresearch/habitat-lab/releases",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: facebookresearch/habitat-lab (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/facebookresearch/habitat-lab",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "GitHub API: facebookresearch/habitat-sim (stars, licence)",
     "url": "https://api.github.com/repos/facebookresearch/habitat-sim",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "habitat-lab README (maintenance notice)",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "habitat-baselines README: Social Navigation and Social Rearrangement sections",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/habitat-baselines/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "habitat-lab LICENSE",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2019",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "habitat-lab commit history (main)",
     "url": "https://github.com/facebookresearch/habitat-lab/commits/main",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "hssd/hssd-hab dataset card",
     "url": "https://huggingface.co/datasets/hssd/hssd-hab",
     "type": "dataset",
     "publisher": "HSSD authors on Hugging Face",
     "date": "2025-02-14",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "ai-habitat/hab3_episodes dataset card",
     "url": "https://huggingface.co/datasets/ai-habitat/hab3_episodes",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2023-11-28",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "ai-habitat/hab3_bench_assets dataset card",
     "url": "https://huggingface.co/datasets/ai-habitat/hab3_bench_assets",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2024-02-22",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "ai-habitat/habitat_humanoids dataset card",
     "url": "https://huggingface.co/datasets/ai-habitat/habitat_humanoids",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2023-10-18",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "ai-habitat/hab_spot_arm dataset card (Spot URDF licence note)",
     "url": "https://huggingface.co/datasets/ai-habitat/hab_spot_arm",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2025-02-14",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "ai-habitat/ycb dataset card",
     "url": "https://huggingface.co/datasets/ai-habitat/ycb",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Hugging Face Hub API record for ai-habitat/hab3_episodes (downloads)",
     "url": "https://huggingface.co/api/datasets/ai-habitat/hab3_episodes?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Hugging Face Hub API record for ai-habitat/habitat_humanoids (downloads)",
     "url": "https://huggingface.co/api/datasets/ai-habitat/habitat_humanoids?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Hugging Face Hub API record for hssd/hssd-hab (downloads, gated flag)",
     "url": "https://huggingface.co/api/datasets/hssd/hssd-hab?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Semantic Scholar API record for arXiv:2310.13724",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2310.13724?fields=title,citationCount,influentialCitationCount,externalIds,venue,year,publicationDate",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Following the Human Thread in Social Navigation (SDA), Table 1",
     "url": "https://arxiv.org/html/2404.11327",
     "type": "paper",
     "publisher": "arXiv; ICLR 2025 (Sapienza University of Rome et al.)",
     "date": "2024-04-17",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "From Cognition to Precognition: A Future-Aware Framework for Social Navigation (Falcon; Social-HM3D, Social-MP3D)",
     "url": "https://arxiv.org/html/2409.13244",
     "type": "paper",
     "publisher": "arXiv; ICRA 2025",
     "date": "2024-09-20",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Robots Ask the Way: Communication-Enabled Social Navigation (Habitat 3.0c)",
     "url": "https://arxiv.org/html/2607.01044",
     "type": "paper",
     "publisher": "arXiv; IROS 2026",
     "date": "2026-07-01",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "PARTNR: A Benchmark for Planning and Reasoning in Embodied Multi-agent Tasks",
     "url": "https://arxiv.org/abs/2411.00081",
     "type": "paper",
     "publisher": "arXiv (Meta FAIR); ICLR 2025",
     "date": "2024-10-31",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "CST-WM: A Causally Structured World Model for Embodied Visual Tracking",
     "url": "https://arxiv.org/html/2609.06302",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-09-05",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "habitat-lab issue #1839: cannot evaluate the released social_nav_latest.pth checkpoint",
     "url": "https://github.com/facebookresearch/habitat-lab/issues/1839",
     "type": "repo",
     "publisher": "facebookresearch/habitat-lab (user report)",
     "date": "2024-03-07",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "habitat-lab issue #2105: object handles missing in provided episodes",
     "url": "https://github.com/facebookresearch/habitat-lab/issues/2105",
     "type": "repo",
     "publisher": "facebookresearch/habitat-lab (user report and community replies)",
     "date": "2024-11-08",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "habitat-lab issue #1684: IndexError while evaluating baseline social nav",
     "url": "https://github.com/facebookresearch/habitat-lab/issues/1684",
     "type": "repo",
     "publisher": "facebookresearch/habitat-lab (user report and contributor reply)",
     "date": "2023-11-14",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "AI Habitat challenge page (redirects to the 2023 HomeRobot OVMM challenge)",
     "url": "https://aihabitat.org/challenge/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "habitat-sim README (maintenance notice)",
     "url": "https://github.com/facebookresearch/habitat-sim/blob/main/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Semantic Scholar API: citations of arXiv:2310.13724 with contexts (342 records)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2310.13724/citations?fields=title,externalIds,year,publicationDate,contexts,intents&limit=100",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Meta Platforms, Inc. Form 10-K for fiscal year 2025 (cover page: principal executive offices)",
     "url": "https://www.sec.gov/Archives/edgar/data/1326801/000162828026025534/meta-12312025x10kars.htm",
     "type": "report",
     "publisher": "Meta Platforms, Inc. (SEC filing)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "facebookresearch/habitat-sim releases",
     "url": "https://github.com/facebookresearch/habitat-sim/releases",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-02-12",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/core-sim-b.json. Changes from the basic entry: leaderboard set to paper-only (was unknown with a 'papers only' display); added the human-in-the-loop fact, evaluation-episode conflict, licence conflict for the humanoid avatars, derived benchmarks, used_by, issues and readings. The basic entry's '1191±3 FPS' figure was not found in the text; we use the intro's 1190 FPS. Checks ran on 2026-10-10 and into early 2026-10-11 local time."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy (the robot's control model) will do on a real robot.",
     "sub": "We found no study that compares Habitat 3.0 results with real robots.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a robot will work with real people.",
     "sub": "Only one study has tested this. It had 30 people and two policies.",
     "basis": [
      "facts.human_in_the_loop"
     ]
    },
    {
     "id": "l3",
     "text": "How a robot copes with natural human behaviour.",
     "sub": "The simulated people can only walk and reach.",
     "basis": [
      "issues.i3"
     ]
    }
   ],
   "validity": []
  },
  {
   "id": "habitat-navigation-challenge",
   "name": "Habitat Navigation Challenge",
   "full_name": "Habitat Navigation Challenge (PointNav, ObjectNav, ImageNav)",
   "aliases": [
    "Habitat Challenge",
    "Habitat PointNav",
    "PointNav-v2",
    "Habitat ObjectNav",
    "HM3D ObjectNav",
    "HM3D-Semantics ObjectNav",
    "InstanceImageNav (Habitat 2023)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Organiser-run scoring of simulated navigation agents (embodied in a robot-like body, Stretch config in 2023) in 3D-scanned buildings.",
   "summary": {
    "text": "The Habitat Navigation Challenge was a yearly contest, run by Meta AI's Habitat team from 2019 to 2023, in which a simulated wheeled robot must reach given coordinates, an object of a named category or the object shown in a photo, inside 3D scans of real buildings it has not seen. The organisers ran each submitted agent on hidden test episodes and ranked entries by SPL, a success rate that also rewards short paths.",
    "sources": [
     "s1",
     "s2",
     "s5"
    ],
    "short": "The Habitat Navigation Challenge was a yearly contest from 2019 to 2023. A simulated robot had to reach a given point or object inside scanned buildings it had not seen before."
   },
   "facts": {
    "kind": {
     "value": "challenge",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Each edition page describes a yearly challenge with deadlines, prizes (2019–2021) and winners announced at CVPR workshops."
    },
    "kind_secondary": {
     "value": [
      "benchmark"
     ],
     "display": "The task rules and episode datasets (PointNav on Gibson, ObjectNav on Matterport3D and HM3D, InstanceImageNav on HM3D) are also used outside the contest, mostly on their public validation splits.",
     "level": "inferred",
     "sources": [
      "s20",
      "s38",
      "s39"
     ],
     "checked": "2026-10-10",
     "note": "Examples: VLFM reports ObjectNav on the HM3D validation split (2,000 episodes); Qwen-RobotNav reports ObjectNav on HM3D v2."
    },
    "version": {
     "value": "2023 edition (last)",
     "display": "Five editions from 2019 to 2023. The tasks, scene datasets and rules changed in almost every edition (see items).",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s3",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2019",
       "display": "PointNav (reach coordinates) on Gibson scenes, RGB and RGB-D tracks. The agent had a perfect GPS+compass, and it slid along walls when it hit them. 2019-04-03 to 2019-05-18.",
       "level": "verified",
       "sources": [
        "s1",
        "s2"
       ]
      },
      {
       "value": "2020",
       "display": "PointNav-v2 on Gibson: no GPS+compass, motion and sensor noise measured on a LoCoBot robot, no sliding, LoCoBot-sized agent. New ObjectNav track on 90 Matterport3D scenes with 21 goal categories, RGB-D camera plus noiseless GPS+compass. 2020-02-24 to 2020-05-31.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "2021",
       "display": "Same two tracks; the PointNav camera was tilted down. 2021-02-17 to 2021-05-31.",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "2022",
       "display": "ObjectNav only, on HM3D-Semantics v0.1: 120 scenes split 80/20/20 and 6 goal categories (chair, couch, potted plant, bed, toilet, tv). PointNav retired, servers left open. 2022-02-14 to 2022-08-31.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "2023",
       "display": "ObjectNav and InstanceImageNav (reach the object shown in a photo) on HM3D-Semantics v0.2: 216 scenes split 145/36/35, same 6 categories, single-floor episodes, Hello Robot Stretch configuration with continuous actions allowed. 2023-03-13 to 2023-05-31.",
       "level": "verified",
       "sources": [
        "s5"
       ]
      }
     ],
     "short": "5 editions, from 2019 to 2023"
    },
    "publishers": {
     "value": [
      "Meta AI (FAIR)",
      "Matterport"
     ],
     "level": "verified",
     "sources": [
      "s5",
      "s17",
      "s22"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Meta AI (FAIR)",
       "display": "Habitat team. EvalAI host team 'FAIR A-STAR (Habitat)'; the challenge site is copyright Meta Platforms. The 2022 and 2023 citation entries list 13 and 16 organisers.",
       "level": "verified",
       "sources": [
        "s4",
        "s5",
        "s17"
       ]
      },
      {
       "value": "Matterport",
       "display": "Released the HM3D scenes used in 2022 and 2023 together with Facebook AI Research.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      }
     ],
     "note": "The challenge pages give organiser names but not affiliations. Co-authors of the task papers come from Georgia Tech, Simon Fraser University, Oregon State and other universities (s29, s32)."
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s5",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Hosted and run by Meta AI's Habitat team, with academic co-organisers. 'consortium' would also be a fair reading."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Meta AI's Habitat team and most named co-organisers are US- or Canada-based."
    },
    "first_release": {
     "value": "2019-04",
     "display": "Habitat Challenge 2019 opened on 2019-04-03 (EvalAI challenge 254).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "short": "April 2019"
    },
    "latest_update": {
     "value": "2023-05",
     "display": "The last edition closed on 2023-05-31. The EvalAI test servers for 2021–2023 stayed open until 2026-01-16. The newest public test-standard entry is dated 2024-05-13 (2022 ObjectNav).",
     "level": "verified",
     "sources": [
      "s5",
      "s17",
      "s48",
      "s11"
     ],
     "checked": "2026-10-10",
     "short": "The last edition ended in May 2023."
    },
    "status": {
     "value": "dormant",
     "display": "No edition after 2023. The starter repository was archived on 2023-10-31, the EvalAI phases ended on 2026-01-16, and Meta stopped active development of Habitat-Lab after v0.3.4 (2026-05-07).",
     "level": "inferred",
     "sources": [
      "s18",
      "s47",
      "s17",
      "s19",
      "s21",
      "s49"
     ],
     "checked": "2026-10-10",
     "note": "aihabitat.org/challenge/ redirects to the 2023 HomeRobot OVMM page and /challenge/2024/ returns HTTP 404 (checked 2026-10-10). The Habitat-Lab README now says the project 'is no longer receiving official active development or maintenance by Meta internal teams'.",
     "short": "Closed. The last edition was in 2023."
    },
    "capability": {
     "value": [
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "object-instance"
     ],
     "display": "Test episodes are in scanned buildings that agents never saw in training. ObjectNav goals are fixed categories, so test objects are new instances of known categories.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s5",
      "s36"
     ],
     "checked": "2026-10-10",
     "note": "InstanceImageNav (2023) samples goal-photo cameras independently of the agent's camera (s5, s36)."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "Habitat-Sim with Habitat-Lab",
     "level": "verified",
     "sources": [
      "s5",
      "s18"
     ],
     "checked": "2026-10-10",
     "short": "Habitat-Sim"
    },
    "embodiment": {
     "value": [
      "mobile-base"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "A simulated wheeled robot that only moves its base. In 2023 it was modelled on the Hello Robot Stretch, a mobile manipulator, but the tasks did not use the arm."
    },
    "robots": {
     "value": "LoCoBot-like agent; Hello Robot Stretch configuration (2023)",
     "display": "2020–2021 PointNav matched a LoCoBot's size, camera and motion noise; ObjectNav matched an Azure Kinect camera; 2023 modelled the Hello Robot Stretch.",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s31"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home",
      "mixed"
     ],
     "display": "Scans of real buildings: Gibson and Matterport3D homes and other buildings; HM3D residential, commercial and civic spaces.",
     "level": "inferred",
     "sources": [
      "s22",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "HM3D README: 'residential, commercial, and civic spaces'. Kadian et al. describe Gibson as apartments, houses, offices, hospitals and gyms."
    },
    "scenes": {
     "value": 216,
     "display": "2019–2021 PointNav: Gibson scenes with Gibson's own splits. 2020–2021 ObjectNav: 90 Matterport3D scenes. 2022: 120 HM3D-Semantics v0.1 scenes (80/20/20). 2023: 216 HM3D-Semantics v0.2 scenes (145/36/35).",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "Up to 216 scanned buildings, in 2023"
    },
    "tasks": {
     "value": 3,
     "display": "Three tasks across editions: PointNav (reach given coordinates), ObjectNav (find any instance of a named category) and InstanceImageNav (reach the object shown in a goal photo)",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "3 tasks, with the goal given as a point, an object category or a photo"
    },
    "objects": {
     "value": 6,
     "display": "ObjectNav goal categories: 21 Matterport3D categories in 2020–2021; 6 categories in 2022–2023 (chair, couch, potted plant, bed, toilet, tv)",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "6 goal categories in 2022 and 2023"
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Entries were ranked by SPL. Tables also show success rate, soft SPL and distance to goal (2023 adds collisions and steps)."
    },
    "metric_detail": {
     "value": "SPL",
     "display": "Success: the agent must stop close enough to the goal. PointNav (2020–2021) required stopping within 0.36 m, twice the agent's radius. ObjectNav requires stopping within 1.0 m of any instance of the target category, at a spot from which the object could be seen by turning or tilting the camera. SPL (success weighted by path length) gives each successful episode the shortest-path length divided by the longer of the agent's path and the shortest path; failed episodes score 0; the result is averaged. A perfect, direct run scores 1.0.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "In ObjectNav the shortest path is measured to the instance closest to the start, so stopping at a farther chair counts as a success but lowers SPL. From 2021 the organisers reserved the right to use other metrics when SPL differences are statistically insignificant.",
     "short": "Ranked by SPL (a success rate that also rewards short paths). Success is also shown."
    },
    "trials": {
     "value": "1,000–2,000 episodes per submission",
     "display": "Hidden test episodes: 1,000 to 2,000 per submission in 2020–2021, 1,000 in 2022–2023, within 24 hours (2020) or 48 hours (2021–2023) on an AWS p2.xlarge instance with a Tesla K80 GPU.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Test-challenge allowed 5 submissions per team in total; test-standard allowed up to 10 a day."
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s7",
      "s11",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboards show single numbers per entry."
    },
    "evaluator": {
     "value": "organiser-run",
     "display": "The organisers. Teams uploaded Docker containers, which EvalAI ran on AWS against hidden test-standard and test-challenge episodes.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Outside the contest, and since the servers closed, papers report their own runs on public validation splits (see issues.i4)."
    },
    "leaderboard": {
     "value": "official",
     "display": "Official EvalAI leaderboards for each edition (challenges 254, 580, 802, 1615 and 1992). All are closed; the 2021–2023 phases ended on 2026-01-16.",
     "level": "verified",
     "sources": [
      "s6",
      "s17",
      "s48"
     ],
     "checked": "2026-10-10",
     "note": "Public 2023 test-challenge tables list two non-baseline ObjectNav teams and one InstanceImageNav team. Only entries that teams chose to make public appear.",
     "short": "Official, on EvalAI. It is now closed."
    },
    "top_score": {
     "value": 0.3661,
     "display": "Best organiser-scored ObjectNav result: SPL 0.3661 with 68.0% success on HM3D v0.1 test-standard (ByteBOT, 2022-08-28). Best PointNav-v2: SPL 0.7446 with 93.15% success (MultiModalVO, 2022-11-04). Editions differ, so numbers across rows are not comparable.",
     "level": "verified",
     "sources": [
      "s11",
      "s9"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2019 PointNav RGB-D",
       "display": "Winner Arnold, SPL 0.948 (test-challenge). Later test-standard best: DD-PPO, SPL 0.9482 (2019-09-23). RGB track winner: Arnold, SPL 0.805.",
       "level": "verified",
       "sources": [
        "s1",
        "s6"
       ]
      },
      {
       "value": "2020 PointNav-v2",
       "display": "Winner OccupancyAnticipation, SPL 0.21, success 0.28. Test-standard best: VO, SPL 0.5248, success 0.7165 (2020-11-14).",
       "level": "verified",
       "sources": [
        "s2",
        "s7"
       ]
      },
      {
       "value": "2021 PointNav-v2",
       "display": "Winner inspir.ai robotics, SPL 0.74, success 0.96. Test-standard best: MultiModalVO (VOT), SPL 0.7446, success 0.9315 (2022-11-04).",
       "level": "verified",
       "sources": [
        "s3",
        "s9"
       ]
      },
      {
       "value": "2020 ObjectNav (Matterport3D)",
       "display": "Winner Arnold (SemExp), SPL 0.10, success 0.25. Test-standard best: THDA, SPL 0.0851, success 0.2018 (2021-06-12).",
       "level": "verified",
       "sources": [
        "s2",
        "s8"
       ]
      },
      {
       "value": "2021 ObjectNav (Matterport3D)",
       "display": "Winner Red Rabbit (6-Act Tether), SPL 0.13, success 0.30. Test-standard best: RIM, SPL 0.1565, success 0.3757 (2023-02-28).",
       "level": "verified",
       "sources": [
        "s3",
        "s10"
       ]
      },
      {
       "value": "2022 ObjectNav (HM3D v0.1)",
       "display": "Winner ByteBOT, SPL 0.3487, success 0.644 (test-challenge). Test-standard best: ByteBOT, SPL 0.3661, success 0.68. Newest entry: AZSU Skip-SCAR, SPL 0.3433, success 0.609 (2024-05-13).",
       "level": "verified",
       "sources": [
        "s4",
        "s12",
        "s11"
       ]
      },
      {
       "value": "2023 ObjectNav (HM3D v0.2)",
       "display": "Winner SkillFusion (AIRI), SPL 0.2712, success 0.533 (test-challenge, 2023-06-05)",
       "level": "verified",
       "sources": [
        "s14",
        "s13",
        "s52"
       ]
      },
      {
       "value": "2023 InstanceImageNav",
       "display": "Winner LQ, SPL 0.0315, success 0.059 (test-challenge, 2023-05-29)",
       "level": "verified",
       "sources": [
        "s16",
        "s15",
        "s52"
       ]
      },
      {
       "value": "Self-reported, after the contest",
       "display": "Papers report higher numbers on validation splits, for example VLFM 52.5% success and SPL 0.304 on HM3D v1 (2023-12) and Qwen-RobotNav-4B 75.6% success on HM3D v2 (2026-06). These runs were not scored by the organisers.",
       "level": "verified",
       "sources": [
        "s38",
        "s39"
       ]
      }
     ],
     "short": "ObjectNav SPL of 0.366 with 68% success, in 2022"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s18",
      "s19",
      "s47"
     ],
     "checked": "2026-10-10",
     "note": "habitat-challenge (archived), Habitat-Lab and Habitat-Sim are all MIT (GitHub licence fields and README)."
    },
    "license_data": {
     "value": "CC-BY-NC-SA-3.0-US",
     "display": "Episode datasets built on Matterport3D or Gibson: CC BY-NC-SA 3.0 US, plus the scene dataset's terms. HM3D-based episode sets (ObjectNav 2022–2023, InstanceImageNav): no licence statement found.",
     "level": "verified",
     "sources": [
      "s19",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "Habitat-Lab README: 'The trained models and the task datasets are considered data derived from the correspondent scene datasets.' DATASETS.md lists the HM3D episode downloads without a licence.",
     "short": "CC BY-NC-SA 3.0 US for the Matterport3D and Gibson episodes"
    },
    "license_assets": {
     "value": "Matterport academic licence (custom)",
     "display": "HM3D, HM3D-Semantics and Matterport3D scenes fall under Matterport's End User License Agreement for Academic Use: non-commercial academic use only. The licence defines models trained on the data as 'derived information' and forbids using it for non-academic purposes. Gibson scenes need a signed licence agreement.",
     "level": "verified",
     "sources": [
      "s22",
      "s23",
      "s24",
      "s25",
      "s26",
      "s27",
      "s28"
     ],
     "checked": "2026-10-10",
     "note": "The Matterport3D Terms of Use PDF contains the same academic-use agreement. The Gibson agreement PDF linked from Habitat-Lab returned HTTP 403 on 2026-10-10 (s28), so its terms were not read.",
     "short": "Matterport terms for academic use only"
    },
    "access": {
     "value": "application",
     "level": "inferred",
     "sources": [
      "s22",
      "s24",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "Scenes are gated. HM3D: request access from Matterport, then use an API token. Matterport3D: sign the terms of use and email them to receive a download script. Gibson: fill in a form with the licence agreement. Episode files and code are open downloads."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s23",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The Matterport academic licence allows only non-commercial academic use of the scenes and of models trained on them; the episode datasets on Matterport3D and Gibson are CC BY-NC-SA. Code is MIT. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Measured twice by teams that include the organisers. PointNav under 2019 rules: Pearson correlation of success across 9 models was 0.18, and 0.844 after tuning the simulator. ObjectNav on 2022 data: the simulation ranking of 4 end-to-end variants was reversed in a real home.",
     "level": "inferred",
     "sources": [
      "s29",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; both studies are verified. Kadian et al. ran the same 9 PointNav models (trained on 72 Gibson scenes) in a scanned replica of a 6.5 m x 10 m lab and on a real LoCoBot: 3 room layouts x 5 episodes x 3 trials, 810 runs in total, about 40.5 hours of robot time. Their Sim2Real Correlation Coefficient (SRCC) is the Pearson correlation of per-model scores. Under 2019 challenge settings, SRCC was 0.603 for SPL and 0.18 for success; nearly all models scored close to 100% success in simulation while real success varied widely. A grid search over simulator settings found sliding off and zero motion noise best, raising SRCC to 0.875 (SPL) and 0.844 (success). The search used the same 9 models and real runs, and no held-out check is reported. The study's agents used GPS+compass from a lidar, so the full 2020 PointNav-v2 rules (no GPS+compass, LoCoBot motion noise) were not measured against a real robot. Gervet et al. compared ObjectNav policies on the 2022 challenge validation split (1,093 single-floor episodes in 20 HM3D scenes) with a Hello Robot Stretch in real homes (see validity). Both studies include challenge organisers (Batra; Chaplot), so neither is independent. No independent replication found (see searched).",
     "short": "Measured by the organisers. The ObjectNav ranking was reversed in a real home."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation."
    },
    "derived_benchmarks": {
     "value": [
      "HM3D-OVON",
      "MultiON",
      "HomeRobot OVMM Challenge"
     ],
     "display": "Later benchmarks and challenges built on the same scenes, tasks or software",
     "level": "verified",
     "sources": [
      "s40",
      "s41",
      "s42"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "HM3D-OVON",
       "display": "2024-09. Open-vocabulary ObjectNav on HM3D-Semantics with 379 categories.",
       "level": "verified",
       "sources": [
        "s40"
       ]
      },
      {
       "value": "MultiON",
       "display": "2020-12. Navigate to an ordered sequence of objects in Habitat.",
       "level": "verified",
       "sources": [
        "s41"
       ]
      },
      {
       "value": "HomeRobot OVMM Challenge",
       "display": "NeurIPS 2023. Open-vocabulary mobile manipulation in Habitat with a real-world counterpart, on the same challenge site.",
       "level": "verified",
       "sources": [
        "s42"
       ]
      }
     ],
     "note": "Not a complete list. VLN-CE and its RxR-Habitat challenge also run in Habitat (separate Atlas entry).",
     "short": "3 later benchmarks built on the same scenes, tasks or software"
    },
    "citations": {
     "value": 2185,
     "display": "No single challenge paper. The Habitat platform paper, cited on every edition page, has 2,185 citations (Semantic Scholar). Task and dataset papers are listed in items.",
     "level": "verified",
     "sources": [
      "s43"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": 2185,
       "display": "Habitat: A Platform for Embodied AI Research (2019), cited as reference 1 on every edition page: 2,185 (251 influential)",
       "level": "verified",
       "sources": [
        "s43"
       ]
      },
      {
       "value": 856,
       "display": "HM3D (2021): 856 (105 influential)",
       "level": "verified",
       "sources": [
        "s44"
       ]
      },
      {
       "value": 191,
       "display": "HM3D-Semantics (2022): 191 (19 influential)",
       "level": "verified",
       "sources": [
        "s45"
       ]
      },
      {
       "value": 222,
       "display": "Navigating to Objects in the Real World (2022): 222 (17 influential)",
       "level": "verified",
       "sources": [
        "s46"
       ]
      }
     ],
     "note": "The count measures use of the Habitat platform, which is wider than use of the challenge.",
     "short": "2,185 for the Habitat platform paper"
    },
    "github_stars": {
     "value": 359,
     "display": "359 stars on facebookresearch/habitat-challenge (archived). Habitat-Lab has 3,154 and Habitat-Sim 3,834.",
     "level": "verified",
     "sources": [
      "s47"
     ],
     "checked": "2026-10-10",
     "short": "359 for the challenge repository"
    },
    "used_by": {
     "value": "ObjectNav submissions: 563 from 27 teams (2020), 400 from 45 teams (2021), 1,022 from 54 teams (2022).",
     "level": "verified",
     "sources": [
      "s32"
     ],
     "checked": "2026-10-10",
     "note": "Counts from the HM3D-Semantics paper (Section 4.4), written by organisers. We found no 2023 submission count.",
     "items": [
      {
       "value": "SemExp (team Arnold)",
       "display": "2020 ObjectNav winner (Chaplot, Gandhi, Gupta, Salakhutdinov)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "OccupancyAnticipation",
       "display": "2020 PointNav winner (Ramakrishnan, Al-Halah, Grauman)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "PIRLNav",
       "display": "Imitation then RL fine-tuning; 65.0% ObjectNav success reported in its abstract; test-standard row 'BadSeed, PIRLNav' 65.4%.",
       "level": "verified",
       "sources": [
        "s37",
        "s11"
       ]
      },
      {
       "value": "ByteBOT",
       "display": "2022 ObjectNav winner (Zhu, Li, Kong); the page gives no affiliation.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "VLFM",
       "display": "Boston Dynamics AI Institute and Georgia Tech, 2023. Zero-shot ObjectNav on HM3D, MP3D and Gibson validation splits; also run on a Spot robot.",
       "level": "verified",
       "sources": [
        "s38"
       ]
      },
      {
       "value": "Qwen-RobotNav",
       "display": "Qwen Team (Alibaba), 2026-06. Reports ObjectNav on MP3D and HM3D v2.",
       "level": "verified",
       "sources": [
        "s39"
       ]
      }
     ],
     "short": "1,022 submissions from 54 teams in 2022"
    },
    "industry_use": {
     "value": [
      "Meta",
      "Samsung",
      "Boston Dynamics AI Institute",
      "Alibaba"
     ],
     "level": "verified",
     "sources": [
      "s5",
      "s2",
      "s38",
      "s39"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Meta",
       "display": "Organiser and host of every edition.",
       "level": "verified",
       "sources": [
        "s5",
        "s17"
       ]
      },
      {
       "value": "Samsung",
       "display": "Samsung Research China – Beijing placed second in 2020 ObjectNav.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Boston Dynamics AI Institute",
       "display": "VLFM reports Habitat ObjectNav results and deploys the method on a Spot robot.",
       "level": "verified",
       "sources": [
        "s38"
       ]
      },
      {
       "value": "Alibaba",
       "display": "Qwen-RobotNav reports Habitat ObjectNav results on HM3D v2.",
       "level": "verified",
       "sources": [
        "s39"
       ]
      }
     ],
     "note": "Team names such as 'inspir.ai robotics' (2021 PointNav winner) and 'ByteBOT' (2022 ObjectNav winner) suggest company teams, but the pages give no affiliations."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "shortcut",
     "title": "Agents exploited wall sliding under the 2019 rules",
     "text": "Kadian et al. found that agents trained under the 2019 settings slid along walls after collisions, which let them take paths through space a real robot cannot pass. Under those settings nearly all 9 tested models scored close to 100% success in simulation, while real-robot success varied widely; the correlation of success between simulation and reality was 0.18. From 2020 the challenge disabled sliding, removed the perfect GPS+compass and added motion noise.",
     "level": "verified",
     "sources": [
      "s29",
      "s2"
     ],
     "status": "addressed",
     "short": "In 2019, agents slid along walls in ways a real robot cannot."
    },
    {
     "id": "i2",
     "type": "saturated",
     "title": "PointNav was solved twice",
     "text": "Under the 2019 rules, DD-PPO 'essentially solves the task' (test-standard SPL 0.9482). The HM3D paper reports that HM3D-trained PointNav agents reach 100% on the Gibson test episodes and suggests retiring that episode set. Under the harder 2020–2021 rules, the 2021 winner reached SPL 0.74 and 96% success, while an agent with perfect GPS+compass can reach at most 0.76 SPL and 99% success in that setting. The organisers considered PointNav-v2 solved and dropped it from 2022.",
     "level": "verified",
     "sources": [
      "s35",
      "s6",
      "s33",
      "s31",
      "s3",
      "s4"
     ],
     "status": "addressed",
     "short": "PointNav (reaching given coordinates) scores reached their upper limit, and the task was retired in 2022."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Simulation rankings can reverse on a real robot",
     "text": "Gervet et al. (Meta AI and partners) scored ObjectNav policies on the 2022 challenge validation split and on a Hello Robot Stretch in real homes. Among 4 end-to-end variants, the one that was best in simulation (77% success) scored 0% in the real home, and the one that was worst in simulation (48%) did best in the real home (30%). Modular methods rose from 81% in simulation to 90% across six real homes. Failures differed: real errors came mostly from depth-sensor noise, while many simulated failures came from scan reconstruction errors.",
     "level": "verified",
     "sources": [
      "s30"
     ],
     "status": "open",
     "short": "In one study in a real home, the end-to-end policy (a single learned network from sensor input to actions) that did best in simulation did worst on the robot."
    },
    {
     "id": "i4",
     "type": "inconsistent-reporting",
     "title": "Numbers come from different datasets and splits",
     "text": "ObjectNav moved from Matterport3D (2020–2021) to HM3D-Semantics v0.1 (2022) and v0.2 (2023), with different scenes and goal categories. Outside the contest, papers report their own validation runs. Qwen-RobotNav's ObjectNav table compares its HM3D v2 numbers with other papers' HM3D v1 numbers. VLFM notes that SemExp was evaluated on a subset of Matterport3D episodes with COCO classes. Before the 2020 challenge, the ObjectNav working group described 'often-inconsistent interpretations' of the task, which the challenge rules standardised.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5",
      "s39",
      "s38",
      "s34"
     ],
     "status": "open",
     "short": "ObjectNav (finding an object of a named category) numbers in papers come from different dataset versions and data splits."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Test servers closed and the software is no longer developed",
     "text": "The 2021–2023 EvalAI phases ended on 2026-01-16, the starter repository was archived on 2023-10-31, and Meta stopped active development of Habitat-Lab after v0.3.4 (2026-05-07). New results can only be self-reported on public splits.",
     "level": "verified",
     "sources": [
      "s17",
      "s48",
      "s47",
      "s19",
      "s21"
     ],
     "status": "open",
     "short": "No test set scored by the organisers is still open."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A high Habitat ObjectNav score says little on its own about success on a real robot. In the one real-home study, run by the organisers' own team, the simulation ranking of end-to-end policies was reversed, while modular methods did better on the robot than in simulation.",
     "basis": [
      "facts.sim_to_real",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "ObjectNav scores are weak evidence of real-robot skill."
    },
    {
     "id": "r2",
     "text": "Compare Habitat navigation numbers only when the task version, scene dataset, split and sensors match. The rules and datasets changed in almost every edition, and since the servers closed, new numbers are self-reported validation runs.",
     "basis": [
      "facts.version",
      "issues.i4",
      "issues.i5",
      "facts.evaluator"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the dataset version, split and sensors before you compare numbers."
    },
    {
     "id": "r3",
     "text": "PointNav under the published Habitat rules is solved. A new PointNav result adds little information about a method.",
     "basis": [
      "issues.i2"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "PointNav results no longer show differences between methods."
    },
    {
     "id": "r4",
     "text": "The 2020 rule changes followed a study that measured how well simulation results matched real-robot results, which is unusual. That study tested agents that had GPS+compass and found that zero motion noise predicted real results best. So the full PointNav-v2 rules were never themselves checked against a real robot.",
     "basis": [
      "facts.sim_to_real",
      "facts.version",
      "sources.s29"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The 2020 rules drew on a real-robot study. The rules as a whole were never checked on a real robot."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How often a policy (the robot's control model) will succeed on a real robot.",
     "sub": "In one test in a real home, the ranking from simulation was reversed.",
     "basis": [
      "issues.i3",
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether models trained on it can be used commercially.",
     "sub": "Matterport's terms allow academic use only.",
     "basis": [
      "facts.license_assets",
      "facts.commercial_use"
     ]
    },
    {
     "id": "l3",
     "text": "How much methods improved from one edition to the next.",
     "sub": "The scenes, sensors and rules changed from year to year.",
     "basis": [
      "facts.version",
      "issues.i4"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "Sim2Real Predictivity (Kadian et al.)",
     "date": "2019-12",
     "by": "authors",
     "method": "The same 9 PointNav models were run in a scanned copy of one lab and on a real LoCoBot robot in that lab. There were 405 runs on each side.",
     "result": "Under the 2019 rules, the Pearson correlation (SRCC) was 0.18 for success and 0.603 for SPL. After the simulator was tuned, it was 0.844 for success and 0.875 for SPL.",
     "authors_view": "low",
     "n_policies": 9,
     "level": "verified",
     "sources": [
      "s29"
     ]
    },
    {
     "id": "v2",
     "name": "Navigating to Objects in the Real World (Gervet et al.)",
     "date": "2022-12",
     "by": "authors",
     "method": "6 ObjectNav policies were scored on the 2022 challenge validation split and in one real home, with 10 episodes. 3 of them were also scored in 6 real homes, with 60 episodes.",
     "result": "The simulation order of 4 end-to-end variants was reversed. They scored 77% to 48% in simulation and 0% to 30% in the real home. The per-episode SRCC was 0.20 to 0.70.",
     "authors_view": "often inversely proportional",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s30"
     ]
    }
   ],
   "searched": [
    {
     "for": "sim_to_real (independent replication)",
     "where": "Web searches on 2026-10-10 for sim-to-real correlation of Habitat PointNav and ObjectNav; Retrospectives on the Embodied AI Workshop; Truong et al. 2022 'Rethinking Sim2Real' (arXiv 2207.10821, same team, own PointNav settings with legged robots, abstract read only); VLFM (real Spot deployment, no paired numbers). No independent paired study found.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data (HM3D episode sets)",
     "where": "Habitat-Lab README and DATASETS.md, HM3D README, HM3D-Semantics page, 2022 and 2023 challenge pages. No licence statement for the HM3D-based ObjectNav and InstanceImageNav episode files.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (Gibson)",
     "where": "Gibson agreement PDF linked from Habitat-Lab (HTTP 403 on 2026-10-10); GibsonEnv data README (licence via a Google form).",
     "date": "2026-10-10"
    },
    {
     "for": "2024 and later editions",
     "where": "aihabitat.org/challenge/2024/ and /2025/ (HTTP 404); challenge menu on aihabitat.org (lists 2019–2023 navigation and 2023 OVMM); EvalAI challenge metadata.",
     "date": "2026-10-10"
    },
    {
     "for": "2023 participation",
     "where": "2023 challenge page (no results section), EvalAI public leaderboards for phases 4704, 4705, 4707 and 4708.",
     "date": "2026-10-10"
    },
    {
     "for": "citations (ObjectNav Revisited, Sim2Real Predictivity)",
     "where": "Semantic Scholar API; repeated HTTP 429 rate-limit responses on 2026-10-10. Counts are added only where the API answered.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Habitat Challenge 2019 (PointGoal, RGB and RGB-D tracks; results)",
     "url": "https://aihabitat.org/challenge/2019/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2019",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Habitat Challenge 2020 (PointNav and ObjectNav; 'New in 2020'; results)",
     "url": "https://aihabitat.org/challenge/2020/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Habitat Challenge 2021 (PointNav and ObjectNav; results)",
     "url": "https://aihabitat.org/challenge/2021/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Habitat Challenge 2022 (ObjectNav on HM3D-Semantics v0.1; results)",
     "url": "https://aihabitat.org/challenge/2022/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Habitat Navigation Challenge 2023 (ObjectNav and InstanceImageNav on HM3D-Semantics v0.2)",
     "url": "https://aihabitat.org/challenge/2023/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "EvalAI leaderboard, Habitat Challenge 2019, PointNav RGB-D test-standard (phase split 839)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/839/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2019",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "EvalAI leaderboard, Habitat Challenge 2020, PointNav test-standard (phase split 1631)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/1631/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "EvalAI leaderboard, Habitat Challenge 2020, ObjectNav test-standard (phase split 1634)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/1634/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "EvalAI leaderboard, Habitat Challenge 2021, PointNav test-standard (phase split 2192)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/2192/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "EvalAI leaderboard, Habitat Challenge 2021, ObjectNav test-standard (phase split 2195)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/2195/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "EvalAI leaderboard, Habitat Challenge 2022, ObjectNav test-standard (phase split 3899)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/3899/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "EvalAI leaderboard, Habitat Challenge 2022, ObjectNav test-challenge (phase split 3900)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/3900/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "EvalAI leaderboard, Habitat Navigation Challenge 2023, ObjectNav test-standard (phase split 4704)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/4704/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "EvalAI leaderboard, Habitat Navigation Challenge 2023, ObjectNav test-challenge (phase split 4705)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/4705/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "EvalAI leaderboard, Habitat Navigation Challenge 2023, InstanceImageNav test-standard (phase split 4707)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/4707/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "EvalAI leaderboard, Habitat Navigation Challenge 2023, InstanceImageNav test-challenge (phase split 4708)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/4708/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI (host team FAIR A-STAR (Habitat))",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "EvalAI API: Habitat Navigation Challenge 2023 phases (end date 2026-01-16, inactive)",
     "url": "https://eval.ai/api/challenges/challenge/1992/challenge_phase",
     "type": "leaderboard",
     "publisher": "EvalAI",
     "date": "2026-01-16",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "facebookresearch/habitat-challenge (starter code; MIT; archived 2023-10-31)",
     "url": "https://github.com/facebookresearch/habitat-challenge",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2023-04-24",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Habitat-Lab README (licence of task datasets; support notice after v0.3.4)",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Habitat-Lab DATASETS.md (PointNav, ObjectNav and InstanceImageNav episode downloads)",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/DATASETS.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Habitat-Lab releases (v0.3.4 on 2026-05-07)",
     "url": "https://github.com/facebookresearch/habitat-lab/releases",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Habitat-Matterport 3D Research Dataset (HM3D) README",
     "url": "https://github.com/matterport/habitat-matterport-3dresearch",
     "type": "repo",
     "publisher": "Matterport",
     "date": "2023-03",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Matterport End User License Agreement for Academic Use of Model Data",
     "url": "https://matterport.com/legal/matterport-end-user-license-agreement-academic-use-model-data",
     "type": "site",
     "publisher": "Matterport",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Matterport3D Terms of Use (PDF linked from Habitat-Lab and VLN-CE)",
     "url": "http://kaldir.vc.in.tum.de/matterport/MP_TOS.pdf",
     "type": "site",
     "publisher": "Matterport",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "HM3D-Semantics dataset page",
     "url": "https://aihabitat.org/datasets/hm3d-semantics/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "HM3D dataset page",
     "url": "https://aihabitat.org/datasets/hm3d/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Gibson Database of Spaces download notes (licence agreement via form)",
     "url": "https://github.com/StanfordVL/GibsonEnv/blob/master/gibson/data/README.md",
     "type": "repo",
     "publisher": "Stanford Vision and Learning Lab",
     "date": "2018",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Gibson Terms of Use agreement PDF linked from Habitat-Lab (returned HTTP 403 on 2026-10-10)",
     "url": "https://storage.googleapis.com/gibson_material/Agreement%20GDS%2006-04-18.pdf",
     "type": "site",
     "publisher": "Stanford (Gibson)",
     "date": "2018-06",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Sim2Real Predictivity: Does Evaluation in Simulation Predict Real-World Performance? (Kadian et al., RA-L 2020; full text v2)",
     "url": "https://arxiv.org/abs/1912.06321",
     "type": "paper",
     "publisher": "IEEE RA-L 2020 (FAIR, Georgia Tech, Oregon State, Simon Fraser University)",
     "date": "2019-12",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Navigating to Objects in the Real World (Gervet et al.; Table 1, Figure 3)",
     "url": "https://arxiv.org/abs/2212.00922",
     "type": "paper",
     "publisher": "arXiv; Science Robotics 2023 (CMU, UC Berkeley, Georgia Tech, Meta AI)",
     "date": "2022-12",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Retrospectives on the Embodied AI Workshop (Section 3.1)",
     "url": "https://arxiv.org/abs/2210.06849",
     "type": "paper",
     "publisher": "arXiv (Embodied AI Workshop organisers)",
     "date": "2022-10",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Habitat-Matterport 3D Semantics Dataset (Section 4.4, Table 5)",
     "url": "https://arxiv.org/abs/2210.05633",
     "type": "paper",
     "publisher": "arXiv (Meta AI, Georgia Tech and others)",
     "date": "2022-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Habitat-Matterport 3D Dataset (HM3D): 1000 Large-scale 3D Environments for Embodied AI",
     "url": "https://arxiv.org/abs/2109.08238",
     "type": "paper",
     "publisher": "NeurIPS 2021 Datasets and Benchmarks",
     "date": "2021-09",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "ObjectNav Revisited: On Evaluation of Embodied Agents Navigating to Objects",
     "url": "https://arxiv.org/abs/2006.13171",
     "type": "paper",
     "publisher": "arXiv (working group report)",
     "date": "2020-06",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "DD-PPO: Learning Near-Perfect PointGoal Navigators from 2.5 Billion Frames",
     "url": "https://arxiv.org/abs/1911.00357",
     "type": "paper",
     "publisher": "ICLR 2020",
     "date": "2019-11",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Instance-Specific Image Goal Navigation: Training Embodied Agents to Find Object Instances",
     "url": "https://arxiv.org/abs/2211.15876",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2022-11",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "PIRLNav: Pretraining with Imitation and RL Finetuning for ObjectNav",
     "url": "https://arxiv.org/abs/2301.07302",
     "type": "paper",
     "publisher": "CVPR 2023",
     "date": "2023-01",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "VLFM: Vision-Language Frontier Maps for Zero-Shot Semantic Navigation (Table I)",
     "url": "https://arxiv.org/abs/2312.03275",
     "type": "paper",
     "publisher": "arXiv (Boston Dynamics AI Institute, Georgia Tech)",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "Qwen-RobotNav Technical Report (Table 4: ObjectNav on MP3D and HM3D)",
     "url": "https://arxiv.org/abs/2606.18112",
     "type": "paper",
     "publisher": "arXiv (Qwen Team, Alibaba)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "HM3D-OVON: A Dataset and Benchmark for Open-Vocabulary Object Goal Navigation",
     "url": "https://arxiv.org/abs/2409.14296",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "MultiON: Benchmarking Semantic Map Memory using Multi-Object Navigation",
     "url": "https://arxiv.org/abs/2012.03912",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2020-12",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "NeurIPS 2023 HomeRobot Open Vocabulary Mobile Manipulation (OVMM) Challenge",
     "url": "https://aihabitat.org/challenge/2023_homerobot_ovmm/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "Semantic Scholar API record for arXiv:1904.01201 (Habitat platform paper)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:1904.01201?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "Semantic Scholar API record for arXiv:2109.08238 (HM3D)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2109.08238?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "Semantic Scholar API record for arXiv:2210.05633 (HM3D-Semantics)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2210.05633?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "Semantic Scholar API record for arXiv:2212.00922 (Navigating to Objects in the Real World)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2212.00922?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "GitHub API: facebookresearch/habitat-challenge, habitat-lab and habitat-sim (stars, archive date)",
     "url": "https://api.github.com/repos/facebookresearch/habitat-challenge",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "EvalAI API: Habitat Challenge 2022 phases (end date 2026-01-16, inactive)",
     "url": "https://eval.ai/api/challenges/challenge/1615/challenge_phase",
     "type": "leaderboard",
     "publisher": "EvalAI",
     "date": "2026-01-16",
     "accessed": "2026-10-10"
    },
    "s52": {
     "title": "Embodied AI Workshop, CVPR 2023 (challenge table with 2023 winners)",
     "url": "https://embodied-ai.org/cvpr2023/",
     "type": "site",
     "publisher": "Embodied AI Workshop organisers",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "AI Habitat challenge index (redirects to the 2023 HomeRobot OVMM page; /challenge/2024/ returns 404)",
     "url": "https://aihabitat.org/challenge/",
     "type": "site",
     "publisher": "Meta AI (AI Habitat)",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Full entry. Re-checked all basic facts. Corrections: the 2019 PointNav winner 'Arnold' scored SPL 0.948 on test-challenge (the later test-standard best is DD-PPO, 0.9482); the Gervet et al. SRCC values are per-episode correlations for single policies, not cross-policy correlations; the ObjectNav sim-vs-real headline (77% to 23%) compares the best simulated end-to-end variant with the variant run at scale, which scored 48% in simulation. Added per-edition results from EvalAI, participation counts, licences for HM3D and Matterport3D (non-commercial, covering trained models), validity entries, and five issues."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "hm3d-ovon",
   "name": "HM3D-OVON",
   "aliases": [
    "OVON",
    "Open-Vocabulary ObjectNav (HM3D)",
    "Habitat-Matterport 3D Open Vocabulary Object Goal Navigation"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed episodes and metrics that score simulated navigation agents (Stretch-like body) searching for objects named in free text.",
   "summary": {
    "text": "Simulation benchmark for finding objects named by free-form text, across 379 categories in 3D-scanned homes.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Georgia Tech (author emails @gatech.edu) and Meta (Abhishek Das 'is with Meta')",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Lead authors at Georgia Tech."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-09 (arXiv v1 2024-09-22)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Repo created 2022-11-09 (GitHub API), before the paper."
    },
    "latest_update": {
     "value": "Repo last commit 2025-05-15 ('updated submodule commit to use reset_metric patch'); HF episodes last modified 2024-09-29",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Commit date via GitHub API; HF date via https://huggingface.co/datasets/nyokoyama/hm3d_ovon"
    },
    "version": {
     "value": "arXiv v1 (no later versions)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "README abstract."
    },
    "embodiment": {
     "value": [
      "mobile-base"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home",
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "379 categories (280 train / 178 eval); 15k+ instances; 181 scenes (145 train / 36 val); '50k' episodes per train scene and '3k' per val scene",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The arXiv HTML renders these as '50 k k' and '3 k k' (math-mode duplication); read as 50k and 3k."
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "Val Unseen: DAgRL+OD SR 37.1±0.2, SPL 19.9±0.3 (best in paper); VLFM SR 35.2, SPL 19.6",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Authors' baselines."
    },
    "leaderboard": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "Papers only (no leaderboard or challenge found)",
     "note": "Checked project page, repo README, EvalAI challenge list (2026-10-10)."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Without a licence file, reuse terms are unstated."
    },
    "license_data": {
     "value": "Episodes: MIT (HF dataset card 'license: mit'); scenes: HM3D under Matterport academic-use EULA",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "HM3D terms: https://github.com/matterport/habitat-matterport-3dresearch"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Paper reports no real-robot experiments; it uses actuation/sensor noise models only. No follow-up measuring sim-vs-real on OVON found."
    },
    "citations": {
     "value": 115,
     "display": "115 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 147,
     "display": "147 (naokiyokoyama/ovon)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Last commit 2025-05; no newer releases."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "HM3D-OVON: A Dataset and Benchmark for Open-Vocabulary Object Goal Navigation",
     "url": "https://arxiv.org/abs/2409.14296",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-09"
    },
    "s2": {
     "title": "HM3D-OVON: A Dataset and Benchmark for Open-Vocabulary Object Goal Navigation (full text)",
     "url": "https://arxiv.org/html/2409.14296v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-09"
    },
    "s3": {
     "title": "HM3D-OVON: A Dataset and Benchmark for Open-Vocabulary Object Goal Navigation",
     "url": "https://naoki.io/portfolio/ovon",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "naokiyokoyama/ovon on GitHub (repository)",
     "url": "https://github.com/naokiyokoyama/ovon",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "nyokoyama/hm3d_ovon on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/nyokoyama/hm3d_ovon",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/ARXIV:2409.14296?fields=citationCount",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s7": {
     "title": "naokiyokoyama/ovon on GitHub (repository)",
     "url": "https://api.github.com/repos/naokiyokoyama/ovon",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "humanoidbench",
   "name": "HumanoidBench",
   "full_name": "HumanoidBench: Simulated Humanoid Benchmark for Whole-Body Locomotion and Manipulation",
   "aliases": [
    "humanoid-bench",
    "HBench"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated humanoid tasks with task rewards and success thresholds produce a score for a whole-body control policy.",
   "summary": {
    "text": "HumanoidBench is a set of 27 simulated tasks in which a Unitree H1 humanoid with two robot hands must walk, balance or handle objects, built at UC Berkeley to test reinforcement learning. A method is scored by the reward its controller collects in the MuJoCo physics simulator, compared with a per-task success threshold; nothing is run on a real robot.",
    "short": "HumanoidBench is a set of 27 simulated tasks for a humanoid robot with two hands. It tests reinforcement learning methods (which learn by trial and error from a reward) and scores them by the reward they collect.",
    "sources": [
     "s1",
     "s4",
     "s6"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents a fixed task suite with reward functions, success thresholds and baseline results."
    },
    "publishers": {
     "value": [
      "UC Berkeley",
      "Yonsei University"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Carmelo Sferrazza, Dun-Ming Huang, Xingyu Lin, Youngwoon Lee, Pieter Abbeel, all at UC Berkeley; Youngwoon Lee also lists Yonsei University. The VLGE claim of a KAIST affiliation is not supported by the paper.",
     "items": [
      {
       "value": "UC Berkeley",
       "display": "All five authors (Robot Learning Lab, per the LICENSE).",
       "level": "verified",
       "sources": [
        "s1",
        "s7"
       ]
      },
      {
       "value": "Yonsei University",
       "display": "Second affiliation of Youngwoon Lee.",
       "level": "verified",
       "sources": [
        "s1",
        "s4"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "From author affiliations."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Lead organisation is UC Berkeley."
    },
    "first_release": {
     "value": "2024-03",
     "display": "arXiv v1 on 2024-03-15; code repository created 2024-03-18. Published at RSS 2024 (Delft, July 2024).",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "RSS DOI 10.15607/RSS.2024.XX.061. arXiv v2 is dated 2024-06-18.",
     "short": "March 2024, at RSS 2024"
    },
    "latest_update": {
     "value": "2025-09",
     "display": "2025-09-18: note added to the LICENSE file. Last code change 2025-05-20 (lighter dependencies).",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The 2025-09 commits only add a BAIR Commons note to the LICENSE. Code merges in April and May 2025 came from FastTD3 authors (rendering switch, print removal).",
     "short": "September 2025. Only a licence note was added."
    },
    "version": {
     "value": "0.2.0",
     "display": "Git tags v0.1.0 and v0.2.0, both published as releases on 2024-07-02. v0.2.0 added the Unitree G1 robot. setup.py says version 0.2.",
     "level": "verified",
     "sources": [
      "s9",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Not on PyPI as far as we checked; installed from source with pip install -e.",
     "short": "0.2.0, from July 2024"
    },
    "status": {
     "value": "dormant",
     "display": "No code change since May 2025. Widely used as a reinforcement-learning test suite.",
     "level": "inferred",
     "sources": [
      "s8",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Last code change 2025-05-20; last commit of any kind 2025-09-18, over a year before the check date. The maintainer's last issue comment is dated 2024-09-24; 26 issues and pull requests are open, including bug reports from 2025. The basic entry said 'maintained'; changed by the one-year rule.",
     "short": "No code changes since May 2025. Still in active use."
    },
    "capability": {
     "value": [
      "locomotion",
      "manipulation",
      "mobile-manipulation",
      "dexterous"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "12 locomotion tasks (walk, stand, run, reach, hurdle, crawl, maze, sit, balance, stair, slide, pole) and 15 whole-body manipulation tasks (push, cabinet, highbar, door, truck, cube, bookshelf, basketball, window, spoon, kitchen, package, powerlift, room, insert). Several, such as door and truck, need walking and handling together."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "Train and test use the same task. Targets and start positions are randomised in some tasks.",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "For example, push and reach use random targets, sit_hard randomises the robot's pose, cube uses random target orientations and basketball throws from random directions. There is no held-out test set; agents learn and are scored in the same environment.",
     "short": "Only targets and start positions are randomised."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1",
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "MuJoCo 3.1.6",
     "display": "MuJoCo (pinned 3.1.6), 0.002 s physics step, control at 50 Hz. MuJoCo MJX used to pre-train low-level reaching skills.",
     "level": "verified",
     "sources": [
      "s1",
      "s10",
      "s11",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Each control step runs 10 physics steps; episodes last up to 1000 control steps (20 s of simulated time). The paper reports over 1,000 frames per second on one CPU for the default model.",
     "short": "MuJoCo 3.1.6"
    },
    "embodiment": {
     "value": [
      "humanoid",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Unitree H1 with two Shadow Hands (simulated)",
     "display": "Main robot: Unitree H1 with two Shadow Hands, 61 actuated joints and 75 degrees of freedom. Also provided: Unitree G1 with three-finger hands, Agility Digit, Robotiq 2F-85 gripper, Unitree H1 hands.",
     "level": "verified",
     "sources": [
      "s1",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The authors removed the Shadow Hands' forearms to make the robot more human-shaped and say 'this is not currently a realistic model'. H1 runs with position control; G1 and Digit are registered with torque control.",
     "short": "Unitree H1 with two Shadow Hands"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "display": "Small task-specific scenes: flat ground, stairs, a maze, a kitchen, a room, a truck, a bookshelf, a basketball hoop.",
     "level": "inferred",
     "sources": [
      "s1",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Read from the task list."
    },
    "tasks": {
     "value": 27,
     "display": "27 tasks: 12 locomotion and 15 manipulation. 31 environments when easy and hard variants count separately.",
     "level": "verified",
     "sources": [
      "s1",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The README's main list has 31 IDs (h1hand-* plus h1strong-highbar_hard-v0). The same tasks also exist without hands (h1-*, 20 IDs) and for the G1 robot (30 IDs).",
     "short": "27 tasks and 31 environments"
    },
    "demonstrations": {
     "value": 0,
     "display": "None. Agents learn from the reward by trial and error.",
     "level": "verified",
     "sources": [
      "s1",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The repository ships pre-trained low-level reaching policies for the hierarchical baseline, not demonstrations.",
     "short": "None. Agents learn from the reward."
    },
    "scoring": {
     "value": [
      "reward"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Episode return (summed reward), shown as learning curves against environment steps, with a per-task success threshold."
    },
    "metric_detail": {
     "value": "episode return vs a success threshold",
     "display": "Each task has a hand-written reward. For walking-type tasks each step's reward is at most 1, so an episode of 1000 steps scores at most 1000. Papers plot this return against training steps. A dashed line marks a fixed per-task threshold, for example 700 for walk, 800 for stand, 1200 for maze, 12000 for reach and 4 for kitchen. The paper says the lines 'qualitatively indicate task success'.",
     "level": "verified",
     "sources": [
      "s1",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Thresholds are the success_bar values in the environment code. Later papers normalise in different ways: return divided by threshold (SimBa, SimbaV2, FlashSAC); (return minus random score) divided by (threshold minus random score) (BRC, EZ-M); or return relative to one baseline (EfficientTDMPC).",
     "short": "Reward compared with a fixed threshold for each task"
    },
    "trials": {
     "value": "3 seeds per method, typical",
     "display": "Original paper: 3 seeds, about 48 hours of training per run (about 2M steps for TD-MPC2, 10M for DreamerV3, SAC and PPO). Common later protocol: 1M environment steps.",
     "level": "verified",
     "sources": [
      "s1",
      "s25",
      "s28",
      "s26"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "SimBa / SimbaV2",
       "display": "14 locomotion tasks without hands (h1-*), 1M environment steps with action repeat 2, 95% intervals; 10 seeds for SAC-based runs in SimBa.",
       "level": "verified",
       "sources": [
        "s24",
        "s25"
       ]
      },
      {
       "value": "BRC and EZ-M",
       "display": "HumanoidBench-Medium (9 no-hand locomotion tasks) and HumanoidBench-Hard (14 tasks with hands), 1M steps, no action repeat, 3 seeds.",
       "level": "verified",
       "sources": [
        "s27",
        "s28"
       ]
      },
      {
       "value": "FastTD3",
       "display": "39 tasks with many parallel simulations, 3 runs; notes that SimbaV2's action repeat of 2 'is unusual for joint position control'.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      }
     ],
     "short": "Usually 3 seeds. The number of training steps varies."
    },
    "uncertainty_reported": {
     "value": "yes",
     "level": "inferred",
     "sources": [
      "s1",
      "s25",
      "s28",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "The original paper shades one standard deviation over 3 seeds; SimbaV2 and EZ-M report 95% intervals; FastTD3 shades standard deviation. With 3 seeds these intervals are wide."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s4",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions. The repository publishes the authors' own training curves as JSON so others can compare without re-running."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s4",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "No results table on the project page or in the README; only the authors' baseline curves in logs/main_results.json."
    },
    "top_score": {
     "value": "no single headline score",
     "display": "Papers report per-task returns or differently normalised averages, so there is no single number. Within 1M training steps, the best methods now pass the success threshold on most simple locomotion tasks; stairs, hurdles, the maze, balance boards and most manipulation tasks remain below it.",
     "level": "inferred",
     "sources": [
      "s16",
      "s25",
      "s28",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Each number below is verified at its source; the counts of tasks above threshold are our arithmetic.",
     "items": [
      {
       "value": "3 of 31",
       "display": "Original baselines, 2024-03: the best of DreamerV3, TD-MPC2, SAC and PPO passes the threshold only on crawl, stand and walk (mean final return over 3 seeds).",
       "level": "inferred",
       "sources": [
        "s16",
        "s1"
       ],
       "note": "Our computation from the authors' logs/main_results.json and the success_bar values in code."
      },
      {
       "value": 0.822,
       "display": "SimbaV2, 2025-02 (ICML 2025): 0.822 return divided by threshold, averaged over 14 no-hand locomotion tasks at 1M steps (update ratio 8). Same table: TD-MPC2 0.710, Simba 0.657, BRO 0.619, SAC 0.279, DreamerV3 0.022.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "7 of 14",
       "display": "EZ-M, 2026-03: on HumanoidBench-Hard (14 tasks with hands, 1M steps) the best of five methods passes the threshold on stand, walk, run, crawl, pole, sit_simple and sit_hard. EZ-M itself passes on 5 (for example walk 919.96 against 700).",
       "level": "inferred",
       "sources": [
        "s28"
       ],
       "note": "Counted by us from EZ-M Appendix B Table 2. EZ-M lists 800 as the h1hand-crawl threshold; the code gives crawl 700."
      },
      {
       "value": "6 of 9",
       "display": "EZ-M, 2026-03: on HumanoidBench-Medium (9 no-hand tasks, 1M steps) the best method passes on 6; stair (best 497.0), hurdle (542.3) and maze (380.8 against 1200) stay below.",
       "level": "inferred",
       "sources": [
        "s28"
       ],
       "note": "Counted by us from EZ-M Appendix B Table 1."
      },
      {
       "value": "39 tasks",
       "display": "FastTD3, 2025-05: 'solves a range of HumanoidBench tasks in under 3 hours on a single A100 GPU' with many parallel simulations; curves for 39 tasks, no single number in the text.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      }
     ],
     "short": "There is no single score. Most simple locomotion tasks have been passed."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, copyright 2024 Robot Learning Lab at UC Berkeley. It also reproduces licences of bundled code: jaxrl_m, DreamerV3 and TD-MPC2 (MIT), purejaxrl (Apache-2.0), MuJoCo (Apache-2.0). GitHub reports 'NOASSERTION' because the file holds several licences."
    },
    "license_data": {
     "value": "not applicable",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "No dataset is shipped. Pre-trained reaching-policy weights sit in the repository under its LICENSE."
    },
    "license_assets": {
     "value": [
      "BSD-3-Clause",
      "Apache-2.0",
      "MIT"
     ],
     "display": "Robot models: Unitree (BSD-3-Clause), Shadow Hand (Apache-2.0), Digit (MIT), Robotiq 2F-85 (BSD-style, ROS-Industrial); robosuite textures (MIT).",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "All in the LICENSE file since the first commit (2024-03-18). This corrects the inventory note that asset licences were not restated."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s6",
      "s4",
      "s33"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub, installed from source. The project page humanoid-bench.github.io works; the README's website link (sferrazza.cc/humanoidbench_site/) returns HTTP 404.",
     "short": "Open. The code is on GitHub."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Code (MIT) and bundled robot models (BSD-3-Clause, Apache-2.0, MIT, BSD-style) are permissive with attribution. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "Simulation only. The main robot, an H1 with forearm-less Shadow Hands, does not exist as built.",
     "level": "inferred",
     "sources": [
      "s1",
      "s32",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "The paper lists sim-to-real transfer as future work. A user asked about sim-to-real support on 2026-03-01 (issue #68) without reply. FastTD3 shows a real Booster T1 robot, but its policy was trained in MuJoCo Playground, not HumanoidBench. See searched.",
     "short": "It has not been checked against real robots."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "citations": {
     "value": 170,
     "display": "170 (Semantic Scholar; 26 influential)",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "short": "170"
    },
    "github_stars": {
     "value": 798,
     "display": "798 stars, 129 forks (carlosferrazza/humanoid-bench)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "798"
    },
    "used_by": {
     "value": "At least 20 of the 170 citing papers report HumanoidBench results, mostly reinforcement-learning algorithm papers. Our count from papers we opened; a lower bound.",
     "level": "inferred",
     "sources": [
      "s21",
      "s24",
      "s25",
      "s26",
      "s27",
      "s28",
      "s29",
      "s30",
      "s31",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "We opened 31 citing papers; about 25 mention HumanoidBench in their experiments.",
     "items": [
      {
       "value": "MuJoCo MPC evaluation",
       "display": "TU Darmstadt and DFKI, 2024-08. Model predictive control on stand, walk and push; critiques the rewards (issues.i1). Its pull request to Google DeepMind's MuJoCo MPC was closed without merging.",
       "level": "verified",
       "sources": [
        "s22",
        "s23"
       ]
      },
      {
       "value": "SimBa",
       "display": "KAIST, Sony AI and others, 2024-10 (ICLR 2025). 14 locomotion tasks without hands.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "SimbaV2",
       "display": "KAIST, Sony AI, UT Austin, 2025-02 (ICML 2025). HBench (14) score 0.822.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "FastTD3",
       "display": "UC Berkeley, 2025-05. 39 tasks; co-authored by HumanoidBench's lead author.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "BRC (Bigger, Regularized, Categorical)",
       "display": "UC Berkeley, University of Warsaw, Nomagic, CMU, 2025-05. Defines HumanoidBench-Medium and -Hard for multi-task learning.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "EZ-M",
       "display": "Texas A&M, MIT, Harvard, 2026-03. Multi-task model-based RL on Medium and Hard.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "FastDSAC",
       "display": "Eastern Institute of Technology (Ningbo) and others, 2026-03.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "FlashSAC",
       "display": "Holiday Robotics, KAIST, KRAFTON and others, 2026-04. 14 locomotion tasks, normalised by thresholds.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "EfficientTDMPC",
       "display": "2026-05. Normalises returns relative to BMPC.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ],
     "short": "At least 20 reinforcement learning papers, up to October 2026"
    },
    "industry_use": {
     "value": [
      "Sony AI",
      "KRAFTON",
      "Holiday Robotics",
      "Nomagic"
     ],
     "level": "verified",
     "sources": [
      "s24",
      "s25",
      "s29",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "All are co-authors of research papers that report HumanoidBench results; no product use found.",
     "items": [
      {
       "value": "Sony AI",
       "display": "SimBa and SimbaV2 co-authors.",
       "level": "verified",
       "sources": [
        "s24",
        "s25"
       ]
      },
      {
       "value": "KRAFTON",
       "display": "SimBa and FlashSAC co-authors.",
       "level": "verified",
       "sources": [
        "s24",
        "s29"
       ]
      },
      {
       "value": "Holiday Robotics",
       "display": "FlashSAC lead affiliation.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "Nomagic",
       "display": "BRC co-author affiliation.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "HBench (14) and HBench-Hard",
      "HumanoidBench-Medium and HumanoidBench-Hard"
     ],
     "display": "Task subsets defined by later papers, not new environments",
     "level": "verified",
     "sources": [
      "s24",
      "s25",
      "s27",
      "s28"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "HBench (14) / HBench-Hard (5)",
       "display": "SimBa and SimbaV2: 14 no-hand locomotion tasks; a hard subset of run, balance-simple, sit-hard, stair and walk.",
       "level": "verified",
       "sources": [
        "s24",
        "s25"
       ]
      },
      {
       "value": "HumanoidBench-Medium / -Hard",
       "display": "BRC and EZ-M: 9 no-hand locomotion tasks; 14 tasks with hands.",
       "level": "verified",
       "sources": [
        "s27",
        "s28"
       ]
      }
     ],
     "short": "Task subsets used in reinforcement learning papers"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Maximising the reward can give unrealistic motion",
     "text": "Meser and others (TU Darmstadt and DFKI, 2024-08) ran model predictive control on stand, walk and push. They found that optimising HumanoidBench's rewards gives undesirable and unrealistic behaviour: in push, the reward drives the robot into unrecoverable postures to reach the box fast. Adding posture and smoothness terms gave higher HumanoidBench scores with steadier motion. The original paper also reports failures such as clinging to the high bar and colliding with hurdles instead of jumping. FastTD3 (2025-05) found in another simulator that one reward function can give a natural gait with one algorithm and an undeployable gait with another.",
     "level": "verified",
     "sources": [
      "s22",
      "s1",
      "s26"
     ],
     "status": "open",
     "counter": {
      "text": "Part of the critique does not match the code. Meser and others say walk and stand episodes last 2 seconds; the code runs up to 1000 control steps of 0.02 s each, which is 20 seconds.",
      "sources": [
       "s11",
       "s13"
      ],
      "short": "The critics say episodes last 2 seconds. The code runs them for up to 20 seconds."
     },
     "note": "The counter is our reading of tasks.py (frame_skip 10, max 1000 steps) and the 0.002 s timestep in the XML.",
     "short": "A high reward can come from postures a real robot should avoid."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "Papers test different versions of the benchmark",
     "text": "The main benchmark uses the H1 with hands (h1hand-*). Many algorithm papers instead use 14 locomotion tasks without hands (SimBa, SimbaV2, FlashSAC), sometimes with an action repeat of 2. BRC and EZ-M use a 9-task no-hand set and a 14-task set with hands. FastTD3 uses 39 tasks with many parallel simulations. Step budgets range from 1M steps to hours of parallel training, and three different normalisations are in use. Scores from different papers are therefore not comparable without checking these settings.",
     "level": "verified",
     "sources": [
      "s24",
      "s25",
      "s27",
      "s28",
      "s26",
      "s29",
      "s31"
     ],
     "status": "open",
     "short": "Papers differ in whether the robot has hands, which tasks they use, how many training steps they allow and how they normalise scores."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Success thresholds are rough",
     "text": "The paper says its dashed success lines 'qualitatively indicate task success'. They are fixed return levels in code, inherited by related tasks (run, crawl, stair, slide and hurdle all use walk's 700). Passing the line does not check that the task was completed in a defined way, and some later papers print different thresholds (EZ-M lists 800 for h1hand-crawl). The repository's published PPO curves also look like fillers: the paper says PPO ran on 4 tasks, and the JSON's 31 PPO entries are copies of 4 distinct curve sets.",
     "level": "inferred",
     "sources": [
      "s1",
      "s12",
      "s28",
      "s16"
     ],
     "status": "open",
     "note": "Thresholds and the PPO quote are verified; the copy finding is our hash comparison of the 31 PPO entries (groups of 17, 8, 5 and 1 tasks).",
     "short": "The success lines are rough levels of total reward. They do not check that the task was done."
    },
    {
     "id": "i4",
     "type": "saturated",
     "title": "Simple locomotion tasks are close to solved",
     "text": "In 2024 the best original baseline passed the threshold on 3 of 31 environments. By 2026, within 1M training steps, at least one published method passes on 6 of 9 no-hand locomotion tasks and 7 of 14 with-hand tasks (EZ-M tables). Stairs, hurdles, the maze, balance boards and reach stay well below, and most manipulation tasks are rarely reported. The suite is not saturated as a whole.",
     "level": "inferred",
     "sources": [
      "s16",
      "s28",
      "s25"
     ],
     "status": "open",
     "note": "Counts are our arithmetic from the authors' logs and EZ-M Appendix B.",
     "short": "Most basic walking tasks are now passed. The harder tasks are not."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "The simulated robot does not exist",
     "text": "The main robot is a Unitree H1 with two Shadow Hands whose forearms were removed; the paper calls this 'not currently a realistic model'. No HumanoidBench policy has been tested on hardware, and a 2026 request for sim-to-real support has no reply.",
     "level": "verified",
     "sources": [
      "s1",
      "s32"
     ],
     "status": "open",
     "short": "The main robot model cannot be built, and nothing was run on hardware."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Bug reports remain open with no reply from the maintainer",
     "text": "Several 2025 reports are unanswered. Issue #40 (2025-04) could not reproduce the authors' window-task results. Issue #60 (2025-10) reports that truck-task packages have no density set. Our reading of truck.xml confirms this, so MuJoCo's default of 1000 kg per cubic metre applies, which makes the smallest package about 24 kg; the separate package task sets a density of 5. The maintainer's last issue comment was on 2024-09-24.",
     "level": "inferred",
     "sources": [
      "s18",
      "s19",
      "s14",
      "s15",
      "s20",
      "s17"
     ],
     "status": "open",
     "note": "Package mass is our arithmetic: a 0.4 x 0.2 x 0.3 m box at 1000 kg per cubic metre.",
     "short": "Bug reports have had no replies, including one about heavy packages in the truck task."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "HumanoidBench tests reinforcement-learning methods in simulation: how fast and how far a method raises the reward. It says little about real humanoid robots, because the robot model is not buildable and nothing has been checked on hardware.",
     "basis": [
      "facts.sim_to_real",
      "issues.i5",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "It is a simulation test for learning methods. It does not test real robots."
    },
    {
     "id": "r2",
     "text": "Read per-task results. Averages hide that walking-type tasks are close to the threshold while stairs, hurdles, the maze and most manipulation tasks are far from it.",
     "basis": [
      "facts.top_score",
      "issues.i4"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Look at the results for each task instead of averages."
    },
    {
     "id": "r3",
     "text": "Before comparing two papers, check whether hands were on, which task subset was used, the step budget, the action repeat and the normalisation.",
     "basis": [
      "issues.i2",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the settings before comparing papers."
    },
    {
     "id": "r4",
     "text": "A high return is not proof of good motion. Reward-maximising behaviour can be jerky or unsafe, so videos or extra measures of smoothness are worth asking for.",
     "basis": [
      "issues.i1",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "High reward does not guarantee good motion."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a controller will do on a real robot.",
     "sub": "HumanoidBench runs only in simulation. Its main robot model cannot be built as a real robot.",
     "basis": [
      "facts.sim_to_real",
      "issues.i5"
     ]
    },
    {
     "id": "l2",
     "text": "Whether the motion is natural and safe.",
     "sub": "A high reward can come from odd postures.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "How results from different papers compare.",
     "sub": "Papers differ in task sets, in whether the robot has hands and in the number of training steps.",
     "basis": [
      "issues.i2"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Paper (arXiv v2, full text), project page, README and all GitHub issues; MuJoCo MPC evaluation (2408.00342); FastTD3 (2505.22642, real robot trained in MuJoCo Playground); FlashSAC (2604.04539); 31 of the 170 Semantic Scholar citing papers opened. No study compares HumanoidBench scores with real-robot results. Validity list left empty.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "Authors' logs/main_results.json; SimbaV2 Table 1; EZ-M Appendix B Tables 1 and 2; FastTD3, BRC, FastDSAC, FlashSAC, WarpSAC and EfficientTDMPC texts. Papers use different subsets and normalisations, so no chart series was made.",
     "date": "2026-10-10"
    },
    {
     "for": "status",
     "where": "Commit history, tags and releases, issue comments by the repository owner (last 2024-09-24), open pull requests.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "LICENSE file (9 sections), its history (third-party notices present since the first commit), README references.",
     "date": "2026-10-10"
    },
    {
     "for": "used_by (MuJoCo MPC integration)",
     "where": "google-deepmind/mujoco_mpc pull request #328 (closed, not merged) and the repository tree (no HumanoidBench tasks).",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "HumanoidBench paper, full text (arXiv v2)",
     "url": "https://arxiv.org/html/2403.10506v2",
     "type": "paper",
     "publisher": "arXiv (UC Berkeley)",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "HumanoidBench arXiv abstract page (v1 2024-03-15, v2 2024-06-18)",
     "url": "https://arxiv.org/abs/2403.10506",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "HumanoidBench, Robotics: Science and Systems XX (RSS 2024) proceedings page, paper 61",
     "url": "https://www.roboticsproceedings.org/rss20/p061.html",
     "type": "paper",
     "publisher": "RSS 2024",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "HumanoidBench project page",
     "url": "https://humanoid-bench.github.io/",
     "type": "site",
     "publisher": "UC Berkeley",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "GitHub API: carlosferrazza/humanoid-bench (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/carlosferrazza/humanoid-bench",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "HumanoidBench GitHub README",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/README.md",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "HumanoidBench LICENSE file (MIT plus third-party licences)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/LICENSE",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2025-09-18",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "HumanoidBench commit history",
     "url": "https://github.com/carlosferrazza/humanoid-bench/commits/main",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2025-09-18",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "HumanoidBench tags and releases (v0.1.0, v0.2.0)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/releases",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024-07-02",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "HumanoidBench setup.py (version 0.2, MuJoCo 3.1.6 pin)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/setup.py",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "HumanoidBench tasks.py (frame_skip 10, max_episode_steps 1000)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/humanoid_bench/tasks.py",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "HumanoidBench environment code (success_bar values per task)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/tree/main/humanoid_bench/envs",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "h1hand_pos_walk.xml (physics timestep 0.002 s)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/humanoid_bench/assets/envs/h1hand_pos_walk.xml",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "truck.xml task asset (package geoms without density)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/humanoid_bench/assets/tasks/truck.xml",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "package.xml task asset (density 5)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/humanoid_bench/assets/tasks/package.xml",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "HumanoidBench logs/main_results.json (authors' training curves for 31 environments)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/blob/main/logs/main_results.json",
     "type": "repo",
     "publisher": "UC Berkeley RLL",
     "date": "2024-04",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "HumanoidBench GitHub issues and pull requests",
     "url": "https://github.com/carlosferrazza/humanoid-bench/issues?q=is%3Aissue",
     "type": "repo",
     "publisher": "UC Berkeley RLL and users",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Issue #40: Unable to reproduce h1hand_window results (no reply)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/issues/40",
     "type": "repo",
     "publisher": "GitHub user",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Issue #60: No density or mass on liftable packages in truck environment (no reply)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/issues/60",
     "type": "repo",
     "publisher": "GitHub user",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "MuJoCo XML reference: geom density default 1000",
     "url": "https://mujoco.readthedocs.io/en/stable/XMLreference.html#body-geom-density",
     "type": "site",
     "publisher": "Google DeepMind (MuJoCo documentation)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Semantic Scholar API record for arXiv:2403.10506",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2403.10506?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "MuJoCo MPC for Humanoid Control: Evaluation on HumanoidBench",
     "url": "https://arxiv.org/abs/2408.00342",
     "type": "paper",
     "publisher": "arXiv (TU Darmstadt, DFKI)",
     "date": "2024-08",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "google-deepmind/mujoco_mpc pull request #328 (HumanoidBench tasks; closed, not merged)",
     "url": "https://github.com/google-deepmind/mujoco_mpc/pull/328",
     "type": "repo",
     "publisher": "Google DeepMind (repository)",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "SimBa: Simplicity Bias for Scaling Up Parameters in Deep Reinforcement Learning (Appendix H.3)",
     "url": "https://arxiv.org/abs/2410.09754",
     "type": "paper",
     "publisher": "arXiv (KAIST, Sony AI, KRAFTON, UT Austin, Coventry); ICLR 2025",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Hyperspherical Normalization for Scalable Deep Reinforcement Learning (SimbaV2; Table 1)",
     "url": "https://arxiv.org/abs/2502.15280",
     "type": "paper",
     "publisher": "arXiv (KAIST, Sony AI, UT Austin); ICML 2025",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "FastTD3: Simple, Fast, and Capable Reinforcement Learning for Humanoid Control",
     "url": "https://arxiv.org/abs/2505.22642",
     "type": "paper",
     "publisher": "arXiv (UC Berkeley)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Bigger, Regularized, Categorical: High-Capacity Value Functions are Efficient Multi-Task Learners (BRC)",
     "url": "https://arxiv.org/abs/2505.23150",
     "type": "paper",
     "publisher": "arXiv (UC Berkeley, University of Warsaw, Nomagic, CMU)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Scaling Tasks, Not Samples: Mastering Humanoid Control through Multi-Task Model-Based RL (EZ-M; Appendix B)",
     "url": "https://arxiv.org/abs/2603.01452",
     "type": "paper",
     "publisher": "arXiv (Texas A&M, MIT, Harvard)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control",
     "url": "https://arxiv.org/abs/2604.04539",
     "type": "paper",
     "publisher": "arXiv (Holiday Robotics, KAIST, KRAFTON and others)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "FastDSAC: Unlocking the Potential of Maximum Entropy RL in High-Dimensional Humanoid Control",
     "url": "https://arxiv.org/abs/2603.12612",
     "type": "paper",
     "publisher": "arXiv (Eastern Institute of Technology, Ningbo, and others)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "EfficientTDMPC: Improved MPC Objectives for Sample-Efficient Continuous Control",
     "url": "https://arxiv.org/abs/2605.16692",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Issue #68: Will HB support sim to real in the future release? (no reply)",
     "url": "https://github.com/carlosferrazza/humanoid-bench/issues/68",
     "type": "repo",
     "publisher": "GitHub user",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "README website link (returns HTTP 404)",
     "url": "https://sferrazza.cc/humanoidbench_site/",
     "type": "site",
     "publisher": "Carmelo Sferrazza",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/academic-c.json. Changes from the basic entry: status set to dormant (no update for over a year); commercial_use set to allowed with asset licences, which the LICENSE file does list (inventory note corrected). New findings: reward critique and its episode-length error, protocol differences across RL papers, partial saturation of locomotion tasks, duplicated PPO curves in the published logs, truck package density bug."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "humantracker",
   "name": "HumanTracker",
   "full_name": "HumanTracker (with HumanScore)",
   "aliases": [
    "HumanTracker",
    "HumanScore"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed motion set, fixed protocol and metrics that score humanoid whole-body tracking policies in simulation.",
   "summary": {
    "text": "Scores humanoid motion trackers in MuJoCo on 153 hours of mocap by success, pose error and a learned preference metric.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Nankai University, Tsinghua University, Galbot, Shanghai Jiao Tong University, Peking University, Shanghai Qi Zhi Institute. Code and data hosted under Galbot's GitHub/HF org (GalaxyGeneralRobotics).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "builder_type set to academic; it is a mixed academic + robot-company effort."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2026-08 (GitHub repo created 2026-08-11; arXiv v1 2026-08-13; HF dataset created 2026-08-14).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-09-28: nine additional released G1 trackers evaluated; 2026-08-22: SONIC 1.1 evaluated; last commit 2026-09-30.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "arXiv v1; repository has no tags or releases.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "locomotion",
      "human-likeness"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [],
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Table 7: Daily 9,739 traj / 89.29 h; Highly Dynamic 2,676 / 11.01 h; Ground 1,640 / 4.59 h; Interaction 10,940 / 47.78 h; total 24,995 trajectories, 27,480,590 frames, 152.67 h; 24 professional performers.",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "HF release contains the test split only (2,500 clips: Daily 974, Highly Dynamic 268, Interaction 1,094, Ground 164) plus 6,000 preference pairs (4,800 train / 1,200 test, 10 GB).",
       "level": "verified",
       "sources": [
        "s4"
       ],
       "note": "Conflict: paper appendix describes released train.json and test.json manifests; HF dataset checked 2026-10-10 has motions/test.json only."
      },
      {
       "value": "Preference data: 6 doctoral researchers annotated 6,000 original pairs, mirrored to 12,000 records. Abstract says '12K motion pairs containing 24K motions'.",
       "level": "verified",
       "sources": [
        "s3"
       ],
       "note": "Internal inconsistency: 12K are mirrored records of 6,000 annotated pairs."
      }
     ]
    },
    "scoring": {
     "value": [
      "success-rate",
      "fidelity",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Official (Results table in the GitHub README, run by the authors: 14 trackers (GMT, TWIST2, SONIC, SONIC 1.1, Humanoid-GPT, ScaleBFM-M/XL, HEFT, GRIT, TeleopIT, MimicLite variants, HoloMotion). No submission process found.)",
     "note": "Classified 'official' because it is a maintained public page beyond the paper; it is organiser-run, not open submission."
    },
    "license_code": {
     "value": "Apache-2.0 (LICENSE file).",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Apache-2.0 (HF dataset card).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Project page footer licence CC BY-SA 4.0 is for the site template, not the data."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (None found. Paper: evaluation uses one 29-DoF humanoid in MuJoCo; results not evidence of hardware robustness; real hardware listed as future work.)",
     "note": "All evaluation in MuJoCo on one 29-DoF humanoid. Paper says results are not evidence of hardware robustness; real hardware is future work. The repo's 'sim2real' backend is a third-party policy runtime that still runs in MuJoCo."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "HumanTracker: Towards Comprehensive and Human-Aligned Motion Tracking Benchmark",
     "url": "https://arxiv.org/abs/2608.13555",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-08"
    },
    "s2": {
     "title": "GalaxyGeneralRobotics/HumanTracker on GitHub (repository)",
     "url": "https://github.com/GalaxyGeneralRobotics/HumanTracker",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "HumanTracker: Towards Comprehensive and Human-Aligned Motion Tracking Benchmark (full text)",
     "url": "https://arxiv.org/html/2608.13555v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-08"
    },
    "s4": {
     "title": "GalaxyGeneralRobotics/HumanTracker on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/GalaxyGeneralRobotics/HumanTracker",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "GalaxyGeneralRobotics/HumanTracker on GitHub (blob)",
     "url": "https://github.com/GalaxyGeneralRobotics/HumanTracker/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "GalaxyGeneralRobotics/HumanTracker on GitHub (file README.md)",
     "url": "https://github.com/GalaxyGeneralRobotics/HumanTracker/blob/main/src/humantracker/eval/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "humenv",
   "name": "HumEnv",
   "full_name": "HumEnv benchmark",
   "aliases": [
    "HumEnv",
    "humenv.bench"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Re-runnable benchmark (fixed reward, goal and tracking tasks with metrics) for policies controlling a physics-simulated humanoid body in 3D. Split out from the Meta Motivo item because it is the reusable artifact. Body is a human avatar, not a robot.",
   "summary": {
    "text": "Simulated SMPL humanoid environment with reward, goal-reaching and motion-tracking test suites for whole-body control policies.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Meta FAIR (same authors as Meta Motivo paper).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-12-16 (tag v0.1.0).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2025-04-22, v0.2.0: fixes to benchmark distance-matrix metric and actuator range. Last push 2025-04-22.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "v0.2.0.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "No commits since 2025-04-22 (about 18 months before check date).",
     "note": "Dormant by activity."
    },
    "scale": {
     "value": "Paper protocol: 45 reward functions, 50 manually selected goal poses, tracking on AMASS test split (990 motions, ~3 h; train 8,902 motions, ~29 h). Repo: 9 configurable reward classes.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "reward",
      "success-rate",
      "fidelity"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (None found; results in the Motivo paper only.)"
    },
    "license_code": {
     "value": "CC BY-NC 4.0 (LICENSE file).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Tracking and MoCap initialisation need SMPL model and AMASS datasets downloaded by the user after registration.",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "AMASS/SMPL licence texts not reviewed here."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (None found.)",
     "note": "Simulated SMPL character only."
    },
    "venue": {
     "value": "sim",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "capability": {
     "value": [
      "locomotion"
     ],
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "facebookresearch/humenv on GitHub (repository)",
     "url": "https://github.com/facebookresearch/humenv",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s2": {
     "title": "facebookresearch/humenv on GitHub (releases)",
     "url": "https://github.com/facebookresearch/humenv/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models (full text)",
     "url": "https://arxiv.org/html/2504.11054v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-04"
    },
    "s4": {
     "title": "facebookresearch/humenv on GitHub (blob)",
     "url": "https://github.com/facebookresearch/humenv/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "facebookresearch/humenv on GitHub (file README.md)",
     "url": "https://github.com/facebookresearch/humenv/blob/main/data_preparation/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models",
     "url": "https://arxiv.org/abs/2504.11054",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-04"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "intphys-2",
   "name": "IntPhys 2",
   "aliases": [
    "IntPhys2"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Intuitive-physics video benchmark from cognitive science; no robot or embodied agent in the videos. Taxonomy-v0: general video-physics benchmarks not built for robots are 'borderline'. Meta released it with V-JEPA 2, a world model also used for robot planning.",
   "summary": {
    "text": "Tests whether video models can tell physically possible from impossible events in Unreal Engine videos.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "All six authors: FAIR at Meta",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Meta HQ; author locations not stated."
    },
    "first_release": {
     "value": "2025-06 (arXiv v1 2025-06-11; Meta blog 2025-06-11)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "HF dataset object created 2025-05-13 (API) but announced 2025-06-11."
    },
    "latest_update": {
     "value": "Repo commits 2025-10-21: 'Update README.md with Held-Out set instructions', dataloader and transforms fixes.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "version": {
     "value": "arXiv v1 only; no release tags",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "Not published at a venue found. OpenReview lists it as 'Submitted to NeurIPS 2025 Datasets and Benchmarks Track' with venueid ending 'Rejected_Submission'.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "OpenReview API search result; the forum page itself was behind a browser check."
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classification on rendered videos; no control loop."
    },
    "capability": {
     "value": [
      "embodied-reasoning",
      "world-modeling"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Tests intuitive physics; predictive world models scored by surprise."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "Debug 5 scenes / 60 videos; Main 253 scenes / 1,012 videos; Held-out 86 scenes / 344 videos.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Same table in paper and HF card."
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "Best MLLM (Gemini 2.5 Flash) slightly above chance except 64% on Easy; best predictive model V-JEPA 2 below 60%; humans near-perfect.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "Official (: Meta 'Physical Reasoning from Video' HF Space (held-out split, 344 videos, scored only there). On 2026-10-10 the Space showed 'Runtime error' (502 when fetching results dataset facebook/pwm_leaderboard_results_public, which is not publicly accessible).)"
    },
    "license_code": {
     "value": "custom: CC-BY-NC-4.0 with added restriction (evaluation only; generative AI uses prohibited)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE.md text; GitHub reports 'Other'."
    },
    "license_data": {
     "value": "custom: CC-BY-NC-4.0 with the same restriction; Unreal Engine / Epic content subject to Epic's terms",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "HF card metadata cc-by-nc-4.0; restriction in README and LICENSE.md; Epic note in paper footnote."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Not built to predict robot performance. Searched paper, Meta blog, repo."
    },
    "citations": {
     "value": 48,
     "display": "48 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 114,
     "display": "114",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Last commit 2025-10; leaderboard currently broken."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "contamination",
     "title": "Main set released with metadata; authors add a held-out set without labels to detect conta",
     "text": "contamination: Main set released with metadata; authors add a held-out set without labels to detect contamination.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "IntPhys 2: Benchmarking Intuitive Physics Understanding In Complex Synthetic Environments (full text)",
     "url": "https://arxiv.org/html/2506.09849v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s2": {
     "title": "IntPhys 2: Benchmarking Intuitive Physics Understanding In Complex Synthetic Environments",
     "url": "https://arxiv.org/abs/2506.09849",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s3": {
     "title": "facebookresearch/IntPhys2 on GitHub (repository)",
     "url": "https://github.com/facebookresearch/IntPhys2",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "https://api2.openreview.net/notes/search?term=IntPhys%202%20intuitive%20physics%20complex%20synthetic",
     "url": "https://api2.openreview.net/notes/search?term=IntPhys%202%20intuitive%20physics%20complex%20synthetic",
     "type": "index",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "facebook/physical_reasoning_leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/facebook/physical_reasoning_leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "facebookresearch/IntPhys2 on GitHub (file LICENSE.md)",
     "url": "https://github.com/facebookresearch/IntPhys2/blob/main/LICENSE.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "facebook/IntPhys2 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/facebook/IntPhys2",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "isaac-lab-arena",
   "name": "Isaac Lab-Arena",
   "aliases": [
    "IsaacLab-Arena",
    "Isaac Lab Arena"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Framework whose job is to run and score robot policies in simulation (success rate, progress); fits kind 'platform'. It ships no fixed benchmark of its own.",
   "summary": {
    "text": "NVIDIA framework for composing simulated robot tasks and running large-scale policy evaluations on Isaac Lab.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "NVIDIA; co-developed with Lightwheel",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "README: evaluation and task layers designed in close collaboration with Lightwheel; built in collaboration with the RoboLab authors. Checked 2026-10-10."
    },
    "builder_type": {
     "value": "platform-vendor",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "NVIDIA-led; Lightwheel's location not checked. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2026-01 (NVIDIA blog 2026-01-05 announces pre-alpha). Earlier: repo created 2025-08-15; first release-notes commit 2025-11-11; release/0.1.0 branch exists.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Commit dates from GitHub API (https://github.com/isaac-sim/IsaacLab-Arena/commits/main/docs/pages/references/release_notes.rst). Checked 2026-10-10."
    },
    "latest_update": {
     "value": "v0.3.1 release notes committed 2026-10-07 (expanded Newton support, placement recording and replay, deformable and cable tasks, ObjectDisappearVariation). Latest commit 2026-10-09.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "version": {
     "value": "v0.3.1; README status 'Alpha Software - Not an Early Access or General Availability Release'",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "manipulation",
      "mobile-manipulation",
      "dexterous"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Docs examples: tabletop and kitchen pick-and-place, door opening (GR1), G1 loco-manipulation box pick-and-place, Kuka Allegro dexterous lift (RL). Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "single-arm",
      "humanoid",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Named: Franka / DROID, Unitree G1, GR1, Kuka Allegro. Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "tabletop",
      "kitchen"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scale": {
     "value": "No fixed task set. Docs list 5 agentic-generation examples, 3 imitation-learning and 2 RL example environments, plus RoboLab, Kitchen Benchmark and Python environment catalogs (sizes not given). v0.3.0 notes: RoboLab catalog and more than 15 pick-and-place environments.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Docs page states Arena is not another benchmark or library of benchmarks. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Task success rate, subtask predicate progress, per-episode records, sensitivity analysis over perturbed factors. Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Blog points to the Hugging Face LeRobot Environment Hub as a place to publish; no Arena leaderboard found. Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0 (LICENSE.md). README notes Isaac Sim, a required dependency, has proprietary components.",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API reports NOASSERTION, likely because LICENSE.md starts with a Markdown heading; the text is Apache 2.0. Checked 2026-10-10."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked in README, docs index, LICENSE.md. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (says partners will publish 'sim-to-real validated' methods, tasks and datasets)",
     "note": "No sim-to-real measurement published for the framework. NVIDIA blog states partners will publish sim-to-real validated methods (a plan, not evidence)."
    },
    "github_stars": {
     "value": 600,
     "display": "600",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API. Checked 2026-10-10."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "isaac-sim/IsaacLab-Arena on GitHub (repository)",
     "url": "https://github.com/isaac-sim/IsaacLab-Arena",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s2": {
     "title": "Simplify Generalist Robot Policy Evaluation in Simulation with NVIDIA Isaac Lab-Arena | NVIDIA Technical Blog",
     "url": "https://developer.nvidia.com/blog/simplify-generalist-robot-policy-evaluation-in-simulation-with-nvidia-isaac-lab-arena",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "isaac-sim/IsaacLab-Arena on GitHub (file release_notes.rst)",
     "url": "https://github.com/isaac-sim/IsaacLab-Arena/blob/main/docs/pages/references/release_notes.rst",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "Welcome to Isaac Lab-Arena! — isaaclab_arena",
     "url": "https://isaac-sim.github.io/IsaacLab-Arena/main/index.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "isaac-sim/IsaacLab-Arena on GitHub (file LICENSE.md)",
     "url": "https://github.com/isaac-sim/IsaacLab-Arena/blob/main/LICENSE.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "libero",
   "name": "LIBERO",
   "full_name": "LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning",
   "aliases": [
    "LIBERO-Spatial",
    "LIBERO-Object",
    "LIBERO-Goal",
    "LIBERO-100",
    "LIBERO-90",
    "LIBERO-10",
    "LIBERO-Long"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "LIBERO is a simulated benchmark of 130 language-instructed manipulation tasks for one Franka Panda arm, built to study lifelong learning and now widely used to report VLA success rates.",
    "sources": [
     "s2",
     "s11",
     "s37"
    ],
    "short": "130 simulated tasks for one robot arm. Many papers use it to report results for vision-language-action (VLA) models."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper calls it a benchmark with fixed task suites, demonstrations and metrics."
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "Also a demonstration dataset, and a procedural task-generation pipeline",
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The pipeline 'can in principle generate infinitely many tasks' (paper abstract). The taxonomy has no value for a task generator, so it is not tagged 'platform'."
    },
    "version": {
     "value": "0.1.0",
     "display": "Upstream package version 0.1.0. No tagged releases. Four suites: LIBERO-Spatial, LIBERO-Object, LIBERO-Goal (10 tasks each) and LIBERO-100 (split into LIBERO-90 and LIBERO-10).",
     "level": "verified",
     "sources": [
      "s13",
      "s14",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "setup.py and the docs both say 0.1.0. The GitHub tags and releases lists are empty. Two other upstream branches exist: 'docs' (last commit 2023-10-13) and 'X-embodiment' (see items). The packages on PyPI are Hugging Face builds, not upstream (see items).",
     "items": [
      {
       "value": "LIBERO-Spatial",
       "display": "10 tasks. Same objects; two identical bowls in different places. Varies spatial layout. Task: put the named bowl on the plate.",
       "level": "verified",
       "sources": [
        "s2",
        "s17"
       ],
       "note": "Paper section 4.2."
      },
      {
       "value": "LIBERO-Object",
       "display": "10 tasks. Same layout; a different object to pick and place in each task. Varies object type.",
       "level": "verified",
       "sources": [
        "s2",
        "s17"
       ],
       "note": "Paper Figure 1 ('Different objects, same layout') and Section 4.2. Tasks are on the floor (BDDL problem LIBERO_Floor_Manipulation)."
      },
      {
       "value": "LIBERO-Goal",
       "display": "10 tasks. Same objects, same layout. Varies the goal (the motion or behaviour asked for).",
       "level": "verified",
       "sources": [
        "s2",
        "s17"
       ]
      },
      {
       "value": "LIBERO-100",
       "display": "100 tasks with mixed objects, layouts and backgrounds. Split into LIBERO-90 (90 short tasks, used for pretraining) and LIBERO-10 (10 long tasks, used for evaluation).",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ],
       "note": "Paper Figure 1 labels LIBERO-100 'Diverse objects, layouts, backgrounds'. Section 4.2 says its tasks 'entail diverse object interactions and versatile motor skills'."
      },
      {
       "value": "LIBERO-10 = LIBERO-Long",
       "display": "Two names for the same 10 long-horizon tasks. The paper says LIBERO-Long; the code says libero_10.",
       "level": "verified",
       "sources": [
        "s2",
        "s7",
        "s17"
       ]
      },
      {
       "value": "Branch X-embodiment (unmerged)",
       "display": "Cross-embodiment code (first commit 2024-08-22) and a robosuite 1.5 update (2025-03-15). Never merged into master.",
       "level": "verified",
       "sources": [
        "s71"
       ]
      },
      {
       "value": "hf-libero 0.1.4 (Hugging Face fork)",
       "display": "PyPI hf-libero, latest 0.1.4 on 2026-06-10, built from huggingface/LIBERO, a GitHub fork of upstream. LeRobot's 'libero' extra requires it. The fork downloads 3D assets at run time from the Hugging Face dataset lerobot/libero-assets. PyPI 'libero' 0.1.1 (2025-11-03) points to huggingface/lerobot-libero, now archived.",
       "level": "verified",
       "sources": [
        "s55",
        "s56",
        "s57",
        "s58",
        "s61"
       ],
       "note": "So 'pip install libero' does not install the upstream code."
      }
     ],
     "short": "0.1.0. There are no tagged releases."
    },
    "publishers": {
     "value": [
      "The University of Texas at Austin",
      "Sony AI",
      "Tsinghua University"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "The University of Texas at Austin",
       "display": "LARG, RPL, and Statistical Learning & AI groups (project site footer). Six of seven authors.",
       "level": "verified",
       "sources": [
        "s2",
        "s4"
       ]
      },
      {
       "value": "Sony AI",
       "display": "Second affiliation of Peter Stone.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Tsinghua University",
       "display": "Affiliation of Chongkai Gao.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "note": "Authors: Bo Liu, Yifeng Zhu, Chongkai Gao, Yihao Feng, Qiang Liu, Yuke Zhu, Peter Stone. The project site's Research page omits Yihao Feng; arXiv, NeurIPS and the README list all seven."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "From author affiliations: two universities; one author also at Sony AI."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Lead organisation is UT Austin."
    },
    "first_release": {
     "value": "2023-06",
     "display": "arXiv v1 on 2023-06-05. Published at NeurIPS 2023, Datasets and Benchmarks Track.",
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub repository was created on 2023-04-08, before the paper. arXiv v2 is dated 2023-10-14.",
     "short": "June 2023, at NeurIPS 2023"
    },
    "latest_update": {
     "value": "2025-05",
     "display": "2025-05-18: one LIBERO-90 demonstration file replaced on the official Hugging Face mirror, with no changelog. Last code commit: 2025-03-15 (Hugging Face download support).",
     "level": "verified",
     "sources": [
      "s51",
      "s15",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Replaced file: libero_90/LIVING_ROOM_SCENE1_pick_up_the_tomato_sauce_and_put_it_in_the_basket_demo.hdf5. Commit message: the original file 'seems to be corrupted'. Size 590,254,080 bytes before, 806,284,784 after (Hub tree at both revisions). Copies made before that date may hold the old file. The mirror was created 2025-03-13; its card was last edited 2025-03-17.",
     "short": "May 2025. One data file was replaced."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper describes 130 language-conditioned manipulation tasks, with a long-horizon suite. Its stated purpose is lifelong learning (knowledge transfer across a stream of tasks), which has no taxonomy value."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "In the common VLA protocol, test tasks are the training tasks. Only the initial object placement differs, drawn from a fixed set per task.",
     "level": "inferred",
     "sources": [
      "s7",
      "s17",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "README: initial states are fixed for benchmarking. OpenVLA keeps 'the same initial environment configurations' as the original benchmark. In the original lifelong protocol, suites isolate shifts in layout, object and goal between tasks, but each new task comes with its own demonstrations, so this is transfer, not zero-shot generalisation.",
     "short": "Only the start positions change"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "robosuite 1.4.0 on MuJoCo",
     "level": "verified",
     "sources": [
      "s2",
      "s12",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Paper: built on robosuite. requirements.txt pins robosuite==1.4.0. Environment code renders with MuJoCo. Tasks are written as BDDL/PDDL files.",
     "short": "robosuite 1.4, built on MuJoCo"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda (simulated)",
     "level": "verified",
     "sources": [
      "s11",
      "s70",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Code default is robots=['Panda']. Tabletop and kitchen problem classes wrap it as MountedPanda; the floor problem (LIBERO-Object) wraps it as OnTheGroundPanda. The paper text does not name the robot; OpenVLA-OFT names it."
    },
    "scene": {
     "value": [
      "tabletop",
      "kitchen",
      "home"
     ],
     "display": "Fixed-base workspaces styled as kitchen, living room and study. LIBERO-Object tasks sit on the floor.",
     "level": "inferred",
     "sources": [
      "s2",
      "s54"
     ],
     "checked": "2026-10-10",
     "note": "Living room and study are mapped to 'home'. These are single workspaces, not whole houses."
    },
    "tasks": {
     "value": 130,
     "display": "130 tasks in four suites (10 + 10 + 10 + 100)",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s54"
     ],
     "checked": "2026-10-10",
     "note": "Repo has 10 + 10 + 10 + 90 + 10 BDDL task files (counted 2026-10-10). Most VLA papers use only the 40 tasks of Spatial, Object, Goal and Long.",
     "short": "130 tasks in 4 suites"
    },
    "scenes": {
     "value": 20,
     "display": "20 scene layouts in LIBERO-100: 10 kitchen, 6 living room, 4 study",
     "level": "inferred",
     "sources": [
      "s2",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "Counted by us from the scene names in the paper's LIBERO-90 task table (Appendix E.3) and the dataset file names. LIBERO-10 reuses scenes from that set. Spatial, Object and Goal use their own layouts and are not counted here.",
     "short": "20 layouts in LIBERO-100"
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No object count in the paper, project site, docs or README. Repo asset folders (counted 2026-10-10): 14 HOPE grocery items, 17 TurboSquid models, 11 scanned-object entries, 19 articulated-object entries. Folder counts are not a clean count of objects used in tasks."
    },
    "demonstrations": {
     "value": 6500,
     "display": "50 human demonstrations per task; 6,500 in total by our arithmetic. The project site says 65,000.",
     "level": "inferred",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT. The paper states 50 trajectories per task, collected by human teleoperation with a 3Dconnexion SpaceMouse. 130 x 50 = 6,500. The project Datasets page states 65,000. OpenVLA also counts 500 demonstrations per 10-task suite, which supports 50 per task.",
     "items": [
      {
       "value": "50 per task",
       "display": "Human teleoperation with a 3Dconnexion SpaceMouse",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Observation content",
       "display": "Workspace and wrist RGB cameras, proprioception, language instruction, PDDL scene description",
       "level": "verified",
       "sources": [
        "s5"
       ]
      },
      {
       "value": "128 x 128 px images",
       "display": "Original image resolution. OpenVLA re-rendered all demonstrations at 256 x 256.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "1,693 episodes (filtered)",
       "display": "OpenVLA replayed demonstrations and dropped failures: 68, 46, 72 and 121 of 500 in Spatial, Object, Goal and Long. The widely used Physical Intelligence copy has 1,693 episodes, 273,465 frames at 10 fps, over 40 tasks.",
       "level": "verified",
       "sources": [
        "s17",
        "s47",
        "s29"
       ],
       "note": "Small source conflict: πRL (Appendix C.1) calls the same set 1,692 demonstrations."
      },
      {
       "value": "about 100.4 GB",
       "display": "Total size of the 130 HDF5 files on the official Hugging Face mirror (decimal GB, 10^9 bytes)",
       "level": "inferred",
       "sources": [
        "s15"
       ],
       "note": "Summed by us from per-file sizes in the Hugging Face file listing: 6.24 + 7.44 + 6.37 + 66.66 + 13.73 GB. Includes the file replaced on 2025-05-18."
      }
     ],
     "short": "50 human demonstrations per task"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Success is a binary goal predicate checked by the simulator. The original lifelong metrics (FWT, NBT, AUC) are built from success rates and have no taxonomy value of their own."
    },
    "metric_detail": {
     "value": "two protocols",
     "display": "Original paper: three lifelong-learning metrics, forward transfer (FWT), negative backward transfer (NBT) and area under the success curve (AUC), computed while tasks arrive one by one. Later VLA papers: train on the demonstrations, then report the success rate per suite and the plain average over Spatial, Object, Goal and Long.",
     "level": "verified",
     "sources": [
      "s2",
      "s17",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "The two protocols answer different questions. The first asks how well a learner keeps and transfers skills over a sequence. The second asks how well one model fits 40 known tasks. Almost all headline numbers today use the second.",
     "short": "Success rate for each suite, averaged across suites"
    },
    "trials": {
     "value": "varies by paper (10 to 50 episodes per task)",
     "level": "verified",
     "sources": [
      "s2",
      "s17",
      "s22",
      "s32",
      "s50",
      "s31",
      "s33",
      "s52"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Fixed initial states: 50 per task",
       "display": "The benchmark loads a fixed set of initial states per task. This caps the common protocol at 50 distinct starts per task, 500 per suite.",
       "level": "inferred",
       "sources": [
        "s53",
        "s52",
        "s29",
        "s22"
       ],
       "note": "The loader reads <task>.pruned_init. We read the array shape (50 x 92) of one libero_spatial file without executing it. πRL states 500 initial states per suite (10 tasks x 50). openpi indexes initial_states[episode_idx] for 50 trials. Not every task file was checked."
      },
      {
       "value": "Original paper",
       "display": "20 rollouts per task, max 600 steps, every 5 epochs; 3 seeds (100, 200, 300)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "OpenVLA convention",
       "display": "500 trials per suite (50 per task) from the fixed initial states; averaged over 3 seeds. Third-person images rotated 180 degrees at train and test time, because the environments rendered upside down on their hardware.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "openpi evaluation script",
       "display": "50 trials per task, seed 7, images resized to 224 px, step limits 220 to 520 by suite. A code comment warns the environment seed changes object positions even with a fixed initial state.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "LeRobot",
       "display": "Docs recommend 10 episodes per task (400 in total) and averaging over 3 seeds. The code default is 50 episodes (EvalConfig.n_episodes).",
       "level": "verified",
       "sources": [
        "s32",
        "s50"
       ]
      },
      {
       "value": "SmolVLA paper",
       "display": "10 trials per task",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "NVIDIA GR00T N1.7 example",
       "display": "200 episodes per suite (20 per task)",
       "level": "verified",
       "sources": [
        "s33"
       ],
       "note": "The README's percentages do not match its own counts in 3 of 4 rows (e.g. 195/200 shown as 97.65%). Its sample eval command uses --n-episodes 10."
      }
     ],
     "short": "10 to 50 per task, depending on the paper"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s17",
      "s37",
      "s32",
      "s66"
     ],
     "checked": "2026-10-10",
     "note": "The original paper reports mean and standard error over 3 seeds. OpenVLA reports standard error. The 2026 audit finds most benchmarks report only an aggregate success rate. Since 2026-09-16, LeRobot's evaluator prints a 95% Wilson interval and success counts next to every success rate."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s4",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions. Each paper runs its own evaluation."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s7",
      "s44"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the project site or in the repo. The former Papers with Code LIBERO page now redirects to Hugging Face Trending Papers (HTTP 302, checked 2026-10-10)."
    },
    "top_score": {
     "value": 99.3,
     "display": "99.3% average over four suites (CORALSimVLA, March 2026). Highest four-suite average without RL fine-tuning that we found. Rows below use different protocols and are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Each score is verified at its source. 'Highest' is our judgement: it covers the 2026 audit's tracker (snapshot 2026-05-21) and the papers we opened. A fresh web search for newer results could not be run on 2026-10-10. All scores are self-reported success rates (%) on Spatial / Object / Goal / Long and their average.",
     "items": [
      {
       "value": 76.5,
       "display": "OpenVLA (7B), 2024-06: 84.7 / 88.4 / 79.2 / 53.7. Reference point for the VLA protocol.",
       "level": "verified",
       "sources": [
        "s17"
       ],
       "note": "Filtered, re-rendered demos; third-person camera only; one policy per suite; 500 trials per suite x 3 seeds.",
       "data": {
        "model": "OpenVLA (7B)",
        "date": "2024-06",
        "suites": [
         84.7,
         88.4,
         79.2,
         53.7
        ],
        "avg": 76.5,
        "rl": false
       }
      },
      {
       "value": 85.5,
       "display": "π0-FAST, 2025-02: 96.4 / 96.8 / 88.6 / 60.2",
       "level": "verified",
       "sources": [
        "s21",
        "s75",
        "s22"
       ],
       "note": "openpi README at its first commit (2025-02-04). Authors note hyperparameters were not tuned. Removed from the README on 2025-09-03. Inferred, not stated by the README: one policy for all four suites (the pi0 configs train on the 40-task physical-intelligence/libero set) and 50 trials per task (eval script default).",
       "data": {
        "model": "π0-FAST",
        "date": "2025-02",
        "suites": [
         96.4,
         96.8,
         88.6,
         60.2
        ],
        "avg": 85.5,
        "rl": false
       }
      },
      {
       "value": 94.15,
       "display": "π0, 2025-02: 96.8 / 98.8 / 95.8 / 85.2",
       "level": "verified",
       "sources": [
        "s21",
        "s75",
        "s22"
       ],
       "note": "Same source and caveats as π0-FAST. Other papers round it to 94.2.",
       "data": {
        "model": "π0",
        "date": "2025-02",
        "suites": [
         96.8,
         98.8,
         95.8,
         85.2
        ],
        "avg": 94.2,
        "rl": false
       }
      },
      {
       "value": 97.1,
       "display": "OpenVLA-OFT (7B), 2025-02: 97.6 / 98.4 / 97.9 / 94.5",
       "level": "verified",
       "sources": [
        "s18"
       ],
       "note": "RSS 2025. Filtered demos; third-person + wrist camera + proprioception; one policy per suite; 500 trials per suite.",
       "data": {
        "model": "OpenVLA-OFT (7B)",
        "date": "2025-02",
        "suites": [
         97.6,
         98.4,
         97.9,
         94.5
        ],
        "avg": 97.1,
        "rl": false
       }
      },
      {
       "value": 96.85,
       "display": "π0.5, 2025-09: 98.8 / 98.2 / 98.0 / 92.4",
       "level": "verified",
       "sources": [
        "s20",
        "s75",
        "s22"
       ],
       "note": "openpi README, checkpoint pi05_libero at 30k steps. Inferred, not stated by the README: one policy for all four suites (pi05_libero config) and 50 trials per task (eval script default). Other papers round it to 96.9.",
       "data": {
        "model": "π0.5",
        "date": "2025-09",
        "suites": [
         98.8,
         98.2,
         98,
         92.4
        ],
        "avg": 96.8,
        "rl": false
       }
      },
      {
       "value": 98.1,
       "display": "X-VLA (0.9B), 2025-10: 98.2 / 98.6 / 97.8 / 97.6",
       "level": "verified",
       "sources": [
        "s24"
       ],
       "note": "Trials per task and whether one policy covers all suites are not stated in the paper.",
       "data": {
        "model": "X-VLA (0.9B)",
        "date": "2025-10",
        "suites": [
         98.2,
         98.6,
         97.8,
         97.6
        ],
        "avg": 98,
        "rl": false
       }
      },
      {
       "value": 99,
       "display": "π0.5 + MoH (3B), 2025-11: 98.8 / 100 / 98.8 / 98.4",
       "level": "verified",
       "sources": [
        "s27"
       ],
       "note": "ICML 2026. One policy on all four suites; 500 trials per suite; single training run. Its own π0.5 baseline differs from openpi's (see issues.i7).",
       "data": {
        "model": "π0.5 + MoH (3B)",
        "date": "2025-11",
        "suites": [
         98.8,
         100,
         98.8,
         98.4
        ],
        "avg": 99,
        "rl": false
       }
      },
      {
       "value": 98.7,
       "display": "Xiaomi-Robotics-0, 2026-02: 98.8 / 100.0 / 98.8 / 97.2",
       "level": "verified",
       "sources": [
        "s30"
       ],
       "note": "Filtered OpenVLA demos; one model on all four suites; OpenVLA evaluation protocol.",
       "data": {
        "model": "Xiaomi-Robotics-0",
        "date": "2026-02",
        "suites": [
         98.8,
         100,
         98.8,
         97.2
        ],
        "avg": 98.7,
        "rl": false
       }
      },
      {
       "value": 98.6,
       "display": "SimVLA, 2026-02: 99.6 / 99.8 / 98.6 / 96.4",
       "level": "verified",
       "sources": [
        "s25",
        "s26"
       ],
       "note": "One policy on all four suites; 'official test episodes'. Size: 0.5B parameters per its abstract; 0.8B per CORAL, by the same team. Internal conflict: Table 2 gives Spatial 99.6 and Goal 98.6; the text gives 99.4 and 98.2; Table 6 'Default settings' gives 99.4 / 99.8 / 98.6 / 96.4. The average is 98.6 in both tables.",
       "data": {
        "model": "SimVLA",
        "date": "2026-02",
        "suites": [
         99.6,
         99.8,
         98.6,
         96.4
        ],
        "avg": 98.6,
        "rl": false
       }
      },
      {
       "value": 99.3,
       "display": "CORAL on SimVLA, 2026-03: 99.6 / 99.8 / 99.0 / 98.8",
       "level": "verified",
       "sources": [
        "s26"
       ],
       "note": "Trains one LoRA expert per task and picks it from the instruction at run time. This differs from a single multitask policy.",
       "data": {
        "model": "CORAL on SimVLA",
        "date": "2026-03",
        "suites": [
         99.6,
         99.8,
         99,
         98.8
        ],
        "avg": 99.3,
        "rl": false
       }
      },
      {
       "value": 99.1,
       "display": "SimpleVLA-RL on OpenVLA-OFT, 2025-09: 99.4 / 99.1 / 99.2 / 98.5",
       "level": "verified",
       "sources": [
        "s28"
       ],
       "note": "RL fine-tuning inside the LIBERO simulator after supervised training on a re-implemented OpenVLA-OFT (see issues.i7). 50 test scenarios per task.",
       "data": {
        "model": "SimpleVLA-RL on OpenVLA-OFT",
        "date": "2025-09",
        "suites": [
         99.4,
         99.1,
         99.2,
         98.5
        ],
        "avg": 99,
        "rl": true
       }
      },
      {
       "value": 98.3,
       "display": "πRL on π0.5 (Flow-Noise), 2025-10: 99.6 / 100 / 99.6 / 94.0",
       "level": "verified",
       "sources": [
        "s29"
       ],
       "note": "Few-shot supervised training (40 trajectories) then online RL in the simulator; 500 initial states per suite.",
       "data": {
        "model": "πRL on π0.5 (Flow-Noise)",
        "date": "2025-10",
        "suites": [
         99.6,
         100,
         99.6,
         94
        ],
        "avg": 98.3,
        "rl": true
       }
      },
      {
       "value": "per-suite best",
       "display": "Best per suite in the 2026 audit's tracker (no RL): Spatial 99.8 (π0 + T-MEE), Object 100.0 (π0.5 + MoH), Goal 99.5 (MoLA), Long 98.8 (CORAL on SimVLA)",
       "level": "verified",
       "sources": [
        "s37"
       ],
       "note": "As compiled by the audit authors; tracker snapshot 2026-05-21."
      }
     ],
     "short": "99.3% average (March 2026)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: MIT, copyright 2023 Lifelong Robot Learning."
    },
    "license_data": {
     "value": [
      "CC-BY-4.0",
      "Apache-2.0"
     ],
     "display": "Conflict: the README says CC BY 4.0; the official Hugging Face dataset card says Apache-2.0. The Hugging Face copy is now the only live official download.",
     "level": "verified",
     "sources": [
      "s7",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT between two primary sources from the same team. The README links the Hugging Face mirror as an official download, and download_utils.py uses it. Derived copies by other groups carry their own labels (items); they do not decide LIBERO's own licence.",
     "items": [
      {
       "value": "MIT",
       "display": "openvla/modified_libero_rlds (OpenVLA's filtered copy)",
       "level": "verified",
       "sources": [
        "s48"
       ]
      },
      {
       "value": "CC-BY-4.0",
       "display": "physical-intelligence/libero (1,693 episodes)",
       "level": "verified",
       "sources": [
        "s47"
       ]
      },
      {
       "value": "Apache-2.0",
       "display": "lerobot/libero (recommended by LeRobot docs)",
       "level": "verified",
       "sources": [
        "s59"
       ]
      },
      {
       "value": "Apache-2.0",
       "display": "HuggingFaceVLA/libero",
       "level": "verified",
       "sources": [
        "s60"
       ]
      }
     ],
     "short": "Sources conflict: CC BY 4.0 or Apache-2.0"
    },
    "license_assets": {
     "value": "unknown",
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The repo ships 3D assets in folders named turbosquid_objects, stable_hope_objects and stable_scanned_objects. The full repo tree has no licence file besides the root MIT LICENSE, which covers 'the Software'. In the paper's NeurIPS checklist, 4(a) 'did you cite the creators' is Yes, 4(b) on asset licences is N/A; the paper text names no asset source. Hugging Face's re-hosted asset copy (lerobot/libero-assets) has no card and no licence. See issues.i11 for likely origins."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s15",
      "s7",
      "s5",
      "s64"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub. Data only from the ungated Hugging Face dataset yifengzhu-hf/LIBERO-datasets. The four original UT Box links (project Datasets page and download_utils.py) return HTTP 404 (checked 2026-10-10). The Datasets page's four 'Best Model Checkpoints' links are empty, so no official checkpoints can be downloaded. No registration needed.",
     "short": "Open. The data is on Hugging Face."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s8",
      "s7",
      "s15",
      "s62",
      "s63"
     ],
     "checked": "2026-10-10",
     "note": "Code (MIT) and data (CC BY 4.0 or Apache-2.0, whichever applies) both allow commercial use with attribution. The bundled third-party 3D assets have no stated licence, and some folder names point to sources with non-commercial or no-redistribution terms (issues.i11). So the whole package is unclear. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Measured once, by a third party (PolaRiS). Pearson r 0.66 to 0.70 over 5 policies; the authors call it poor. By rank violations (MMRV), one checkpoint did nearly as well as PolaRiS (0.04 vs 0.03).",
     "level": "inferred",
     "sources": [
      "s43",
      "s67"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; the PolaRiS numbers are verified. Design: 5 DROID-trained policies (π0.5, π0, π0 at 100k steps, π0-FAST, PaliGemma binning) were fine-tuned on LIBERO-90 and run 50 times on each of 90 tasks (4,500 runs per policy). Real-world scores come from the base policies: 20 rollouts per policy in each of 6 real scenes, 3 at UW and 3 at Princeton (Figure 6 legend; the text says 'two institutions'). Real tasks are not LIBERO replicas and are scored by a human on a 0 to 1 progress rubric, while LIBERO scores binary success. So this tests whether LIBERO predicts real-world ranking across tasks. Results (Figure 7): Pearson r 0.66 / 0.70 / 0.66 and MMRV 0.19 / 0.04 / 0.15 for LIBERO checkpoints at 1k / 10k / 50k steps; PolaRiS r 0.90, MMRV 0.03. Nearly all policies scored 90 to 95% on LIBERO but spread widely in the real world. The text says the best-correlated checkpoint is reported; Figure 7 shows all three and Figure 6 plots the 10k one. No confidence interval on r, which rests on 5 points. The PolaRiS authors propose a competing method, and Physical Intelligence co-authors also built the policies tested. Separately, VLA-REPLICA asserts without measurement that simulation benchmarks such as LIBERO give overly optimistic real-world estimates ('claimed'-type evidence). No study by the LIBERO authors and no independent replication found (see searched).",
     "short": "Compared once with real robots. Correlation r = 0.66 to 0.70."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "LIBERO-PRO",
      "LIBERO-Plus",
      "LIBERO-X",
      "LangGap",
      "LIBERO-Para",
      "Libero-V",
      "LIBERO-Safety",
      "LIBERO-RECOVER"
     ],
     "display": "Stress tests built on LIBERO tasks or scenes",
     "level": "verified",
     "sources": [
      "s34",
      "s35",
      "s69",
      "s68",
      "s39",
      "s40",
      "s42",
      "s41"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "LIBERO-PRO",
       "display": "2025-10 (v2 2026-05). Perturbs objects, positions, instructions and tasks. See issues.i3.",
       "level": "verified",
       "sources": [
        "s34",
        "s77"
       ],
       "note": "HarnessPAI (2026-09), a code harness around π0.5, reports 96.5% on its Swap and Task perturbations (three suites). It evolves one program per task on 15 of the 50 test seeds."
      },
      {
       "value": "LIBERO-Plus",
       "display": "2025-10; CVPR 2026. Seven perturbation factors; 10,030-task test set. See issues.i4. LeRobot integrates it.",
       "level": "verified",
       "sources": [
        "s35",
        "s36",
        "s65"
       ]
      },
      {
       "value": "LIBERO-X",
       "display": "2026-02. Hierarchical perturbation levels plus a new teleoperated training set.",
       "level": "verified",
       "sources": [
        "s69"
       ]
      },
      {
       "value": "LangGap",
       "display": "2026-02. Varies instruction meaning under a fixed tabletop layout.",
       "level": "verified",
       "sources": [
        "s68"
       ]
      },
      {
       "value": "LIBERO-Para",
       "display": "2026-03; EMNLP 2026. Paraphrased instructions on LIBERO-Goal. See issues.i5.",
       "level": "verified",
       "sources": [
        "s39"
       ]
      },
      {
       "value": "Libero-V",
       "display": "CVPR 2026. Viewpoint, lighting, texture and noise perturbations drawn from LIBERO-Plus.",
       "level": "verified",
       "sources": [
        "s40"
       ]
      },
      {
       "value": "LIBERO-Safety",
       "display": "2026-06; ECCV 2026 per arXiv comment. Physical and semantic safety.",
       "level": "verified",
       "sources": [
        "s42"
       ]
      },
      {
       "value": "LIBERO-RECOVER",
       "display": "2026-09. Failure recovery.",
       "level": "verified",
       "sources": [
        "s41"
       ]
      }
     ],
     "note": "Not a complete list. Each has its own record scope; none has a published sim-to-real study that we found.",
     "short": "8 harder test sets built on LIBERO"
    },
    "citations": {
     "value": 1885,
     "display": "1,885 (Semantic Scholar; 487 influential)",
     "level": "verified",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Re-read 2026-10-10 after several rate-limited attempts.",
     "short": "1,885"
    },
    "github_stars": {
     "value": 2405,
     "display": "2,405 stars, 511 forks (Lifelong-Robot-Learning/LIBERO)",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "short": "2,405"
    },
    "dataset_downloads": {
     "value": 43194,
     "display": "43,194 (Hub 'downloads' field), 309,185 all time, 67 likes: official Hugging Face mirror",
     "level": "verified",
     "sources": [
      "s76"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API. We did not check the time window behind the 'downloads' field. Derived copies are counted separately (e.g. physical-intelligence/libero: 37,924, 93 likes).",
     "short": "43,194 on Hugging Face"
    },
    "used_by": {
     "value": "At least 551 papers reported LIBERO results by 2026-05-21, per the 2026 audit's tracker. 79 of them have a first arXiv date in March 2026.",
     "level": "verified",
     "sources": [
      "s37"
     ],
     "checked": "2026-10-10",
     "note": "The audit's own measurement: 551 of 1,053 candidate rows classified as reporting results. The authors call the counts lower bounds. Rows without an arXiv ID are only left out of monthly counts, so the 551 may include non-arXiv papers.",
     "items": [
      {
       "value": "OpenVLA",
       "display": "Stanford and others, 2024-06. Its protocol (filtered data, 500 trials per suite) is cited as 'the standard evaluation protocol' by later reports.",
       "level": "verified",
       "sources": [
        "s17",
        "s30"
       ],
       "note": "That it set the convention is our reading of later citations, e.g. Xiaomi-Robotics-0."
      },
      {
       "value": "OpenVLA-OFT",
       "display": "Stanford, 2025-02 (RSS 2025)",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "π0, π0-FAST, π0.5",
       "display": "Physical Intelligence, results in the openpi repo, 2025",
       "level": "verified",
       "sources": [
        "s19",
        "s20",
        "s21"
       ]
      },
      {
       "value": "SmolVLA",
       "display": "Hugging Face, 2025-06",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "X-VLA",
       "display": "2025-10",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "GR00T N1.7",
       "display": "NVIDIA, 2026 (repo example with results)",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "Xiaomi-Robotics-0",
       "display": "Xiaomi, 2026-02",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "SimpleVLA-RL, πRL",
       "display": "RL fine-tuning studies, 2025",
       "level": "verified",
       "sources": [
        "s28",
        "s29"
       ]
      },
      {
       "value": "LeRobot π0.5 reproduction",
       "display": "Hugging Face, 2026: 97.5 average with 10 episodes per task",
       "level": "verified",
       "sources": [
        "s32"
       ]
      }
     ],
     "short": "At least 551 papers (May 2026)"
    },
    "industry_use": {
     "value": [
      "Physical Intelligence",
      "Hugging Face",
      "NVIDIA",
      "Xiaomi"
     ],
     "level": "verified",
     "sources": [
      "s23",
      "s32",
      "s65",
      "s57",
      "s33",
      "s30"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Physical Intelligence",
       "display": "openpi uses LIBERO as its main fine-tuning example and publishes a π0.5-LIBERO checkpoint.",
       "level": "verified",
       "sources": [
        "s23",
        "s20"
       ]
      },
      {
       "value": "Hugging Face",
       "display": "LeRobot ships LIBERO and LIBERO-plus environments, docs and eval commands. Hugging Face maintains the hf-libero fork. SmolVLA reports LIBERO.",
       "level": "verified",
       "sources": [
        "s32",
        "s65",
        "s55",
        "s57",
        "s31"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Isaac-GR00T repo has a LIBERO fine-tuning example with GR00T N1.7 results.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "Xiaomi",
       "display": "The Xiaomi-Robotics-0 report gives LIBERO first among its simulation results (98.7% average).",
       "level": "verified",
       "sources": [
        "s30"
       ]
      }
     ]
    },
    "status": {
     "value": "dormant",
     "display": "Upstream code unchanged since 2025-03-15; last data fix 2025-05-18. A Hugging Face fork is maintained. Use is very active.",
     "level": "inferred",
     "sources": [
      "s10",
      "s49",
      "s51",
      "s55",
      "s12",
      "s72",
      "s73",
      "s74"
     ],
     "checked": "2026-10-10",
     "note": "Last merged pull request 2025-01-03. 108 open items (24 PRs, 84 issues); community PRs from October 2026 are unmerged. Last comment by a repo collaborator: 2025-08-06 (issue #99, Python >= 3.9 and new GPUs). requirements.txt pins robosuite 1.4.0 (released 2022-12-01); robosuite on PyPI is 1.5.2. PR #146 for robosuite 1.5.2 / MuJoCo 3.x is open. The hf-libero fork (0.1.4, 2026-06-10) keeps LIBERO installable for LeRobot users. For current use, see facts.used_by.",
     "short": "No code changes since March 2025. Still widely used."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "saturated",
     "title": "Top scores are close to 100%",
     "text": "At least five papers since October 2025 report four-suite averages of 98.1 to 99.3%. Per-suite bests are 99.5 to 100% on Spatial, Object and Goal, and 98.8% on Long. Several later papers call LIBERO saturated.",
     "level": "verified",
     "sources": [
      "s24",
      "s27",
      "s30",
      "s25",
      "s26",
      "s37",
      "s41",
      "s42",
      "s46"
     ],
     "status": "open",
     "short": "Since late 2025, the top average scores have been 98% to 99%. The best scores on single suites reach 99.5% to 100%."
    },
    {
     "id": "i2",
     "type": "shortcut",
     "title": "A small model given only a task number scores close to the best",
     "text": "A 0.09B probe (DINOv2 encoder + MLP) gets a task ID instead of the instruction. It scored 99.0 / 100.0 / 98.8 / 92.4 on Spatial / Object / Goal / Long, within about one point of the best published result on three suites. It was trained and tuned per suite, and these are the best of several checkpoints scored on the test suite. With no checkpoint selection, its mean was 95.1%. The audit says this shows a high LIBERO score is not on its own evidence of general skill. It does not claim high-scoring policies lack skill.",
     "level": "verified",
     "sources": [
      "s37"
     ],
     "status": "open",
     "short": "A model with 0.09 billion parameters was given only a task number and no instruction. It scored within about one point of the best models on three suites."
    },
    {
     "id": "i3",
     "type": "shortcut",
     "title": "High-scoring models fail when object positions or tasks change",
     "text": "LIBERO-PRO perturbs objects, positions, instructions and tasks. Its abstract says models above 90% fall to 0.0% in its generalised setting. The drop depends on the perturbation: position and task changes push success to near zero, while object and instruction-meaning changes barely lower it. OpenVLA and π0 fail once an object moves more than 0.2 units; π0.5 holds to about 0.4 units and keeps 0.38 on LIBERO-Goal under position change. Nonsense instructions leave trajectories nearly unchanged. The authors read this as rote memorisation.",
     "level": "verified",
     "sources": [
      "s34",
      "s25"
     ],
     "status": "contested",
     "counter": {
      "text": "The 2026 audit accepts the drops but disputes the label. It says LIBERO-PRO and LIBERO-Plus change inputs outside the training distribution, so a drop may show weak generalisation rather than overfitting. When it redrew initial states inside the distribution (10,000 rollouts per policy), success moved by under one point (main text: Spatial Forcing −0.62, SimVLA +0.30, LeRobot π0.5 −0.14; its appendix table prints the opposite signs). It notes rollout noise leaves each sign uncertain.",
      "sources": [
       "s37"
      ],
      "short": "The 2026 audit agrees that scores drop, but it describes the cause as weak generalisation. When the audit resampled start positions within the training range, scores changed by less than one point."
     },
     "note": "LIBERO-PRO's results table is an image in the HTML version and was not read. Per-perturbation numbers come from SimVLA's Table 3 reproduction (e.g. OpenVLA and π0.5: Obj and Sem 81 to 98%, Pos 0 to 38%, Task 0 to 1%).",
     "short": "In LIBERO-PRO, models that score above 90% on LIBERO drop to 0% when object positions or tasks change. The LIBERO-PRO authors say the models memorised the tasks."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Scores drop sharply after small camera or start-pose changes",
     "text": "LIBERO-Plus (CVPR 2026) perturbs seven factors. In its single-factor analysis, success drops from about 95% to below 30% under modest camera-viewpoint or robot-initial-state changes. Its released 10,030-task test set dropped tasks that all or most baseline models solved, so its scores are not on the same footing as standard LIBERO scores.",
     "level": "verified",
     "sources": [
      "s35",
      "s36"
     ],
     "status": "open",
     "mitigation": {
      "text": "A CVPR 2026 paper (Sun Yat-sen University) agrees models are brittle to new viewpoints and traces it to visual representations. Its one-shot adaptation uses one human demonstration per task. On Libero-V, a 4K-parameter module raised viewpoint success from 48.5% to 87.1%, and a 4.7M-parameter one reached 90.8%.",
      "sources": [
       "s40"
      ]
     },
     "short": "In LIBERO-Plus, success falls from about 95% to below 30% after small changes to the camera or to the robot's start pose."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Models rely on the exact wording of the instruction",
     "text": "LIBERO-Plus's text says models largely ignore instructions. Its own blank-instruction test shows success falls sharply for most models when the instruction is removed, to about 0 to 10% on LIBERO-Goal for all six models; only OpenVLA-OFT on Object is unchanged. When the target object in the instruction is swapped, success drops to near zero. LIBERO-PRO finds near-identical trajectories under nonsense instructions. LIBERO-Para (EMNLP 2026) finds paraphrases cut success by 22 to 52 points across seven VLA set-ups, mostly through object synonyms; its authors read this as surface-level matching that disrupts task identification.",
     "level": "verified",
     "sources": [
      "s35",
      "s34",
      "s39"
     ],
     "status": "open",
     "note": "Blank-instruction values are read by us from LIBERO-Plus Figure 3(a) bar heights (inferred). LIBERO-Para uses LIBERO-Goal because there the instruction is the only task cue; results may differ on other suites.",
     "short": "Without the instruction, success on LIBERO-Goal falls to 0% to 10%. When the instruction is reworded, success falls by 22 to 52 points."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Most claimed improvements are not shown to be statistically significant",
     "text": "Of 789 previous-best-to-new comparisons on LIBERO (Spatial, Object and Goal pooled), 19.8% are provably significant at the 5% level from public scores. The rest show no improvement, are provably not significant, or cannot be tested because only averages are published. Each comparison uses the previous best as stated in the new paper.",
     "level": "verified",
     "sources": [
      "s37"
     ],
     "status": "open",
     "mitigation": {
      "text": "Since 2026-09-16, LeRobot's evaluator reports success counts and a 95% Wilson interval for every success rate. Papers still mostly publish averages.",
      "sources": [
       "s66",
       "s32",
       "s37"
      ]
     },
     "short": "Only 19.8% of 789 claimed improvements on LIBERO can be shown to be statistically significant from the published numbers."
    },
    {
     "id": "i7",
     "type": "inconsistent-reporting",
     "title": "The same model gets different scores in different papers",
     "text": "π0: 94.15 (openpi, 50 trials per task by script default), 86.0 (SmolVLA paper, 10 trials per task), 96.8 (T-MEE paper's own run). π0.5: 96.85 (openpi), 97.5 (LeRobot, 10 episodes per task), 97.7 (MoH paper's run). OpenVLA-OFT: 97.1 (its paper) vs 91.0 (SimpleVLA-RL's re-implementation with one camera and no proprioception; πRL reprints it labelled only 'OpenVLA-OFT'). SimVLA's own text and Table 2 disagree on two suites.",
     "level": "verified",
     "sources": [
      "s21",
      "s22",
      "s31",
      "s46",
      "s20",
      "s32",
      "s27",
      "s18",
      "s28",
      "s29",
      "s25"
     ],
     "status": "open",
     "note": "Values such as 94.2 and 96.9 in other papers are roundings of the openpi numbers, not separate runs. T-MEE orders its columns Spatial / Goal / Object / Long.",
     "short": "Different papers report the score of the same model, π0, as 86.0, 94.15 and 96.8."
    },
    {
     "id": "i8",
     "type": "protocol-variance",
     "title": "Papers test in different ways",
     "text": "Papers differ in trials per task (10, 20 or 50), training data (original or the filtered 1,693 episodes), one policy per suite vs one for all suites vs one expert per task, averages over three or four suites, and RL fine-tuning inside the test simulator. There is no validation split, so picking the best checkpoint on the test suite is allowed; for its own probe, the audit found this adds 0.6 to 4.4 points. Smaller settings matter too: the openpi script notes the environment seed moves objects even with fixed initial states; LeRobot says soft and hard resets give slightly different results and asks authors to pin the dataset revision; SimVLA finds single training choices can outweigh architecture changes.",
     "level": "verified",
     "sources": [
      "s17",
      "s32",
      "s31",
      "s33",
      "s26",
      "s45",
      "s28",
      "s37",
      "s22",
      "s25"
     ],
     "status": "open",
     "short": "Papers run 10, 20 or 50 trials per task, train on different data and average over different sets of suites."
    },
    {
     "id": "i9",
     "type": "protocol-variance",
     "title": "Results change on different computers",
     "text": "With policy, seed and initial state fixed, changing only the CPU made the simulator state diverge in 10 of 10 LIBERO tasks for OpenVLA-OFT; in 5 of 10 the difference reached images and actions. Changing only the GPU also caused divergence. The audit argues bitwise determinism should not be the only reproducibility test, and finds aggregate moves small.",
     "level": "verified",
     "sources": [
      "s37"
     ],
     "status": "open",
     "short": "In a test of 10 tasks, changing only the computer's CPU changed the simulation results in all 10."
    },
    {
     "id": "i10",
     "type": "shortcut",
     "title": "The scene often shows which task to do",
     "text": "LIBERO draws instructions from a fixed set, one per task, so a task ID can replace language (issues.i2). LangGap says LIBERO assigns only one task per layout. LIBERO-Para says all LIBERO-Goal tasks start from one initial state, so there the instruction is the only cue. The two claims conflict for Goal, which the LIBERO paper describes as same objects and layout with different goals.",
     "level": "verified",
     "sources": [
      "s37",
      "s68",
      "s39",
      "s2"
     ],
     "status": "open",
     "short": "Each layout is used for only one task, so a policy can tell the task from the scene. Sources disagree about whether this is true for the Goal suite."
    },
    {
     "id": "i11",
     "type": "other",
     "title": "The included 3D models have no stated licence",
     "text": "The repo and Hugging Face's re-hosted copy ship 3D assets with no licence file. Folder names suggest third-party origins. stable_hope_objects holds 14 grocery items; NVIDIA's HOPE set is 28 toy grocery items with 3D meshes under CC BY-NC-SA 4.0. turbosquid_objects holds 17 models; TurboSquid's Royalty Free License bars giving away model files outside a permitted Creation. stable_scanned_objects (11 items) has unchecked origin.",
     "level": "inferred",
     "sources": [
      "s54",
      "s61",
      "s62",
      "s63",
      "s2"
     ],
     "status": "open",
     "note": "The link between folders and sources is inferred from names only. The paper names no asset source. Not legal advice.",
     "short": "The 3D models that come with LIBERO have no licence file. Some of them appear to come from collections with restricted licences."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A high LIBERO score shows that a policy can learn these simulated tasks from their demonstrations. It does not show general manipulation skill. It is weak evidence about real-robot performance, because the only paired study found a correlation of r = 0.66 to 0.70 over five policies.",
     "basis": [
      "facts.generalisation",
      "facts.sim_to_real",
      "issues.i2",
      "issues.i3",
      "issues.i4",
      "issues.i10"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A high score shows that a policy learned these simulated tasks. It says little about how the policy will do on real robots."
    },
    {
     "id": "r2",
     "text": "Do not rank models by differences of one or two points. At the top of the leaderboard, such differences are often smaller than the variation between random seeds and the differences between the test setups that papers use.",
     "basis": [
      "facts.top_score",
      "issues.i1",
      "issues.i6",
      "issues.i7",
      "issues.i8"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Do not rank models by differences of one or two points at the top."
    },
    {
     "id": "r3",
     "text": "The harder variants, such as LIBERO-PRO and LIBERO-Plus, show failures that the standard score does not show. Read their scores together with the standard score. None of the variants has been compared with real-robot results, and models can be tuned to them as well.",
     "basis": [
      "facts.derived_benchmarks",
      "issues.i3",
      "issues.i4",
      "issues.i5"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Read the scores on the harder variants together with the standard score."
    },
    {
     "id": "r4",
     "text": "LIBERO was designed to measure how a policy carries knowledge across a stream of new tasks (lifelong learning). In the papers we checked, the headline numbers measure success on 40 known tasks instead. Check which protocol a number comes from.",
     "basis": [
      "facts.metric_detail",
      "facts.trials",
      "facts.top_score"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Most papers now report success on 40 known tasks. Check which protocol a number comes from."
    },
    {
     "id": "r5",
     "text": "LIBERO is still useful for fast iteration, debugging and comparison with your own earlier runs. It is open, cheap to run and widely reproduced. The authors of the 2026 audit say the same about the benchmarks they audited.",
     "basis": [
      "facts.access",
      "facts.license_code",
      "facts.used_by",
      "sources.s37"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "LIBERO is still useful for quick, low-cost testing during development."
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "LIBERO paper (arXiv 2306.03310v2, full text): no real-robot experiments. Project site, docs and README: none. OpenVLA (2406.09246) and OpenVLA-OFT (2502.19645): real-robot and LIBERO results reported separately. RoboArena (2506.18123): cites LIBERO only. VLA-REPLICA (2605.20774): asserts LIBERO-type benchmarks overestimate real performance, no measurement. 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (2606.10366): related work only. 2026 audit (2606.04233): calls a sim-vs-real ranking test impractical and does not run one. References in LIBERO-PRO, LIBERO-Plus, LIBERO-Para, LIBERO-Safety, LIBERO-RECOVER and RoboVerse also followed. The drafting pass ran web searches (queries not logged). A fresh web search could not be run on 2026-10-10 (shared search budget used up); re-run before publication. Only PolaRiS (2512.16881) reports a paired measurement.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "Repo LICENSE, README licence table, full repo git tree (1,367 entries; no other licence files), paper NeurIPS checklist items 4(a) and 4(b), paper full text for asset creators (none named), project site, docs, Hugging Face mirror card, lerobot/libero-assets (no card).",
     "date": "2026-10-10"
    },
    {
     "for": "objects",
     "where": "Paper text and appendix, project main and Datasets pages, docs Overview and Datasets pages, README.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project site (main, Datasets, Research pages), GitHub README, papers with code URL /sota/robot-manipulation-on-libero (redirects to Hugging Face Trending Papers), web search for a LIBERO leaderboard.",
     "date": "2026-10-10"
    },
    {
     "for": "version",
     "where": "GitHub tags and releases lists (both empty), setup.py, docs header.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score (X-VLA trials)",
     "where": "X-VLA paper full text: no count of trials or episodes for LIBERO.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score (anything above 99.3% four-suite average)",
     "where": "Web searches for 2026 LIBERO averages of 99.4% and above; audit tracker per-suite bests. None found; HarnessPAI's 98.1% averages only three suites. A fresh web search on 2026-10-10 could not run (search budget used up); nothing after the audit's 2026-05-21 snapshot was re-searched beyond papers already cited.",
     "date": "2026-10-10"
    },
    {
     "for": "demonstrations (original control or recording frequency)",
     "where": "LIBERO paper full text and README: no Hz or frequency stated. Derived copies use 10 fps.",
     "date": "2026-10-10"
    },
    {
     "for": "issues.i3 (LIBERO-PRO per-perturbation table)",
     "where": "LIBERO-PRO v2 HTML: the results table is an image; text gives only qualitative findings and the π0.5 0.38 value. Used SimVLA Table 3 instead.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (asset origins)",
     "where": "HOPE repo tree and README: no per-object name list to match LIBERO's 14 folders.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2306.03310",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "LIBERO paper, full text v2",
     "url": "https://arxiv.org/pdf/2306.03310v2",
     "type": "paper",
     "publisher": "arXiv (NeurIPS 2023 Datasets and Benchmarks camera-ready text)",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "LIBERO, NeurIPS 2023 proceedings page",
     "url": "https://proceedings.neurips.cc/paper_files/paper/2023/hash/8c3c666820ea055a77726d66fc7d447f-Abstract-Datasets_and_Benchmarks.html",
     "type": "paper",
     "publisher": "NeurIPS 2023 Datasets and Benchmarks Track",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "LIBERO project site, main page",
     "url": "https://libero-project.github.io/main.html",
     "type": "site",
     "publisher": "UT Austin LARG, RPL and Statistical Learning & AI groups",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "LIBERO project site, Datasets page",
     "url": "https://libero-project.github.io/datasets",
     "type": "site",
     "publisher": "LIBERO team",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "LIBERO project site, Research page",
     "url": "https://libero-project.github.io/research",
     "type": "site",
     "publisher": "LIBERO team",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "LIBERO GitHub README",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/README.md",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "LIBERO LICENSE file",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/LICENSE",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "GitHub API: Lifelong-Robot-Learning/LIBERO (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/Lifelong-Robot-Learning/LIBERO",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "LIBERO commit history, tags and releases",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/commits/master",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-03-15",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "LIBERO environment wrapper source (robot and renderer)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/libero/libero/envs/env_wrapper.py",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "LIBERO requirements.txt",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/requirements.txt",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2024-12-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "LIBERO setup.py",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/setup.py",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "LIBERO documentation: Datasets",
     "url": "https://lifelong-robot-learning.github.io/LIBERO/html/algo_data/datasets.html",
     "type": "site",
     "publisher": "LIBERO team",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "LIBERO datasets on Hugging Face (official mirror linked from README), card and file listing",
     "url": "https://huggingface.co/datasets/yifengzhu-hf/LIBERO-datasets",
     "type": "repo",
     "publisher": "Yifeng Zhu (co-author)",
     "date": "2025-03 (created); 2025-05-18 (last data change)",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Semantic Scholar API record for arXiv:2306.03310",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2306.03310?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "OpenVLA: An Open-Source Vision-Language-Action Model (Appendix E, LIBERO)",
     "url": "https://arxiv.org/abs/2406.09246",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success (OpenVLA-OFT)",
     "url": "https://arxiv.org/abs/2502.19645",
     "type": "paper",
     "publisher": "RSS 2025",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2501.09747",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence et al.)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "openpi LIBERO example README (current: π0.5 results)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/examples/libero/README.md",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "openpi LIBERO example README at first commit (π0 and π0-FAST results)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/231a1cf7/examples/libero/README.md",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025-02-04",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "openpi LIBERO evaluation script",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/examples/libero/main.py",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "openpi README (LIBERO fine-tuning example, π0.5-LIBERO checkpoint)",
     "url": "https://github.com/Physical-Intelligence/openpi",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "X-VLA: Soft-Prompted Transformer as Scalable Cross-Embodiment Vision-Language-Action Model",
     "url": "https://arxiv.org/abs/2510.10274",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "SimVLA: A Simple VLA Baseline for Robotic Manipulation",
     "url": "https://arxiv.org/abs/2602.18224",
     "type": "paper",
     "publisher": "arXiv (Frontier Robotics)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "CORAL: Scalable Multi-Task Robot Learning via LoRA Experts",
     "url": "https://arxiv.org/abs/2603.09298",
     "type": "paper",
     "publisher": "arXiv (Frontier Robotics)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Mixture of Horizons in Action Chunking (MoH)",
     "url": "https://arxiv.org/abs/2511.19433",
     "type": "paper",
     "publisher": "ICML 2026",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "SimpleVLA-RL: Scaling VLA Training via Reinforcement Learning",
     "url": "https://arxiv.org/abs/2509.09674",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "πRL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2510.25889",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Xiaomi-Robotics-0: An Open-Sourced Vision-Language-Action Model with Real-Time Execution",
     "url": "https://arxiv.org/abs/2602.12684",
     "type": "paper",
     "publisher": "Xiaomi Robotics",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "SmolVLA: A vision-language-action model for affordable and efficient robotics",
     "url": "https://arxiv.org/abs/2506.01844",
     "type": "paper",
     "publisher": "Hugging Face",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "LeRobot documentation: LIBERO",
     "url": "https://github.com/huggingface/lerobot/blob/main/docs/source/libero.mdx",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Isaac-GR00T LIBERO example and results (GR00T N1.7)",
     "url": "https://github.com/NVIDIA/Isaac-GR00T/blob/main/examples/LIBERO/README.md",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "LIBERO-PRO: Towards Robust and Fair Evaluation of Vision-Language-Action Models Beyond Memorization (v2, 2026-05-25)",
     "url": "https://arxiv.org/html/2510.03827",
     "type": "paper",
     "publisher": "arXiv (HUST, Tsinghua, WUT, Lehigh)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "LIBERO-Plus: In-depth Robustness Analysis of Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2510.13626",
     "type": "paper",
     "publisher": "arXiv (Fudan, Tongji, Shanghai Innovation Institute, NUS)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "LIBERO-Plus: A Progressive Robustness Benchmark for Visual-Language-Action Models (CVPR 2026 poster page)",
     "url": "https://cvpr.thecvf.com/virtual/2026/poster/38735",
     "type": "paper",
     "publisher": "CVPR 2026",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation?",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne); CoRL 2026 per project page",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "Manipulation Benchmark Audit project page",
     "url": "https://ripl.github.io/manipulation_benchmark_audit/",
     "type": "site",
     "publisher": "TTIC RIPL (lists CoRL 2026 and IROS 2026 RGMCW Workshop)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "LIBERO-Para: A Diagnostic Benchmark and Metrics for Paraphrase Robustness in VLA Models",
     "url": "https://arxiv.org/abs/2603.28301",
     "type": "paper",
     "publisher": "EMNLP 2026 (per arXiv comment)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "VLA Models Are More Generalizable Than You Think: Revisiting Physical and Spatial Modeling",
     "url": "https://openaccess.thecvf.com/content/CVPR2026/papers/Li_VLA_Models_Are_More_Generalizable_Than_You_Think_Revisiting_Physical_CVPR_2026_paper.pdf",
     "type": "paper",
     "publisher": "CVPR 2026 (Sun Yat-sen University)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "LIBERO-RECOVER: Beyond Task Success Towards Failure Recovery in Robotic Manipulation Models",
     "url": "https://arxiv.org/abs/2609.05178",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "LIBERO-Safety: A Comprehensive Benchmark for Physical and Semantic Safety in Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2606.23686",
     "type": "paper",
     "publisher": "ECCV 2026 (per arXiv comment)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (v2; Section 5.2, Figures 6 and 7, Appendix C.1)",
     "url": "https://arxiv.org/html/2512.16881v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "Former Papers with Code LIBERO leaderboard URL (redirects to Hugging Face Trending Papers)",
     "url": "https://paperswithcode.com/sota/robot-manipulation-on-libero",
     "type": "secondary",
     "publisher": "Papers with Code / Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "HarnessPAI: An Evolving Harness for Physical AI (three-suite average)",
     "url": "https://arxiv.org/abs/2609.29166",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "Reshaping Action Error Distributions for Reliable Vision-Language-Action Models (T-MEE)",
     "url": "https://arxiv.org/abs/2602.04228",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "physical-intelligence/libero dataset card (1,693 episodes, 40 tasks)",
     "url": "https://huggingface.co/datasets/physical-intelligence/libero",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "openvla/modified_libero_rlds dataset card",
     "url": "https://huggingface.co/datasets/openvla/modified_libero_rlds",
     "type": "repo",
     "publisher": "OpenVLA team",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "LIBERO pull requests and issues",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/pulls",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "LeRobot EvalConfig defaults (n_episodes)",
     "url": "https://github.com/huggingface/lerobot/blob/main/src/lerobot/configs/default.py",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s51": {
     "title": "Hugging Face commit list for yifengzhu-hf/LIBERO-datasets (incl. 2025-05-18 file replacement)",
     "url": "https://huggingface.co/api/datasets/yifengzhu-hf/LIBERO-datasets/commits/main",
     "type": "repo",
     "publisher": "Yifeng Zhu (co-author)",
     "date": "2025-05-18",
     "accessed": "2026-10-10"
    },
    "s52": {
     "title": "LIBERO fixed initial-state file (example: libero_spatial, .pruned_init)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/tree/master/libero/libero/init_files/libero_spatial",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s53": {
     "title": "LIBERO benchmark loader (get_task_init_states reads <task>.pruned_init)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/libero/libero/benchmark/__init__.py",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s54": {
     "title": "LIBERO repo tree: BDDL task files and asset folders",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/tree/master/libero/libero",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s55": {
     "title": "PyPI: hf-libero",
     "url": "https://pypi.org/project/hf-libero/",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-06-10",
     "accessed": "2026-10-10"
    },
    "s56": {
     "title": "PyPI: libero",
     "url": "https://pypi.org/project/libero/",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2025-11-03",
     "accessed": "2026-10-10"
    },
    "s57": {
     "title": "huggingface/LIBERO (GitHub fork of upstream; README 'Assets' section)",
     "url": "https://github.com/huggingface/LIBERO",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s58": {
     "title": "LeRobot pyproject.toml ('libero' extra requires hf-libero)",
     "url": "https://github.com/huggingface/lerobot/blob/main/pyproject.toml",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s59": {
     "title": "lerobot/libero dataset card",
     "url": "https://huggingface.co/datasets/lerobot/libero",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s60": {
     "title": "HuggingFaceVLA/libero dataset card",
     "url": "https://huggingface.co/datasets/HuggingFaceVLA/libero",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s61": {
     "title": "lerobot/libero-assets (Hugging Face dataset; no card)",
     "url": "https://huggingface.co/datasets/lerobot/libero-assets",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s62": {
     "title": "NVIDIA HOPE dataset README (licence section)",
     "url": "https://github.com/swtyree/hope-dataset",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s63": {
     "title": "TurboSquid Royalty Free License",
     "url": "https://blog.turbosquid.com/royalty-free-license/",
     "type": "site",
     "publisher": "TurboSquid",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s64": {
     "title": "LIBERO download_utils.py (UT Box URLs, Hugging Face repo id)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/libero/libero/utils/download_utils.py",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s65": {
     "title": "LeRobot documentation: LIBERO-plus",
     "url": "https://github.com/huggingface/lerobot/blob/main/docs/source/libero_plus.mdx",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s66": {
     "title": "LeRobot commit: report success counts and 95% Wilson intervals (#4628)",
     "url": "https://github.com/huggingface/lerobot/commits/main/docs/source/libero.mdx",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-09-16",
     "accessed": "2026-10-10"
    },
    "s67": {
     "title": "VLA-REPLICA: A Low-Cost, Reproducible Benchmark for Real-World Evaluation of VLA Models",
     "url": "https://arxiv.org/abs/2605.20774",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s68": {
     "title": "LangGap: Diagnosing and Closing the Language Gap in Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2603.00592",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s69": {
     "title": "LIBERO-X: Robustness Litmus for Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2602.06556",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s70": {
     "title": "LIBERO problem classes (MountedPanda / OnTheGroundPanda wrappers)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/tree/master/libero/libero/envs/problems",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s71": {
     "title": "LIBERO branch X-embodiment (unmerged)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/tree/X-embodiment",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-03-15",
     "accessed": "2026-10-10"
    },
    "s72": {
     "title": "LIBERO pull request #146: Support robosuite 1.5.2 / MuJoCo 3.x (open)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/pull/146",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning (community PR)",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s73": {
     "title": "LIBERO issue #99: compatibility with Python >= 3.9 and new GPUs",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/issues/99",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s74": {
     "title": "PyPI: robosuite (release history)",
     "url": "https://pypi.org/project/robosuite/",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s75": {
     "title": "openpi training configs (pi0_libero, pi05_libero use physical-intelligence/libero)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/src/openpi/training/config.py",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s76": {
     "title": "Hugging Face Hub API record for yifengzhu-hf/LIBERO-datasets (downloads, likes)",
     "url": "https://huggingface.co/api/datasets/yifengzhu-hf/LIBERO-datasets?expand[]=downloads&expand[]=likes&expand[]=downloadsAllTime",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s77": {
     "title": "HarnessPAI full text (Section 5.2.3, Figure 13: LIBERO-PRO protocol and results)",
     "url": "https://arxiv.org/pdf/2609.29166",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-09",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created from primary sources (full depth). Re-verified the VLGE prior claims for LIBERO."
    },
    {
     "date": "2026-10-10",
     "change": "Integrated three verifier reports and re-checked every non-confirmed item at the primary source. Fixed: latest_update (data file replaced 2025-05-18), access (UT Box links dead; no checkpoints), LeRobot trial default (docs 10, code 50), robot wrappers, licence sources, used_by wording (lower bound), i1 per-suite bests, Xiaomi wording, openpi protocol notes marked inferred, sim_to_real level to inferred with caveats. Softened i3, i4, i5 and readings r1, r3, r4; added r5. Added i10 (scene identifies task), i11 (asset licences), derived_benchmarks, dataset_downloads, initial-state count, hf-libero fork, Wilson intervals. Kept against verifiers: LIBERO-100 'backgrounds' (paper Figure 1 says so); UW and Princeton as real sites (PolaRiS Figure 6 legend names them); SimVLA size now shows both 0.5B (its abstract) and 0.8B (CORAL). Rejected a verifier case for i7: 96.85 vs 96.9 and 94.15 vs 94.2 are roundings of the same openpi runs. Not re-done: open web search for newer top scores and other sim-to-real studies (search budget used up)."
    },
    {
     "date": "2026-10-10",
     "change": "Published to the Atlas with short display texts. The full texts are unchanged."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy will do on a real robot.",
     "sub": "Only one study has compared LIBERO scores with real-robot results. It found a correlation of r = 0.66 to 0.70 over five policies, and its authors call this poor.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy handles new objects or new layouts.",
     "sub": "The test tasks are the same tasks the policy was trained on. Only the start positions change.",
     "basis": [
      "facts.generalisation"
     ]
    },
    {
     "id": "l3",
     "text": "Whether a policy understands the instruction.",
     "sub": "Studies show that policies can identify the task from the scene or from a task number, without reading the instruction.",
     "basis": [
      "issues.i5",
      "issues.i2"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "PolaRiS",
     "date": "2025-12",
     "by": "independent",
     "method": "The same 5 policies were scored on LIBERO-90 and on real robots. The real-robot tasks were different from the LIBERO tasks.",
     "result": "Correlation r = 0.66 to 0.70",
     "authors_view": "poor",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s43"
     ]
    }
   ]
  },
  {
   "id": "locomujoco",
   "name": "LocoMuJoCo",
   "aliases": [
    "loco-mujoco"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated locomotion tasks with datasets and per-task metrics score imitation-learning policies for legged and humanoid bodies.",
   "summary": {
    "text": "Imitation-learning benchmark for locomotion: humanoid, quadruped and human-body models imitate motion-capture data in MuJoCo.",
    "sources": [
     "s2"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Al-Hafez, Zhao, Peters, Tateo: Intelligent Autonomous Systems Group and Locomotion Laboratory, TU Darmstadt; German Research Center for AI (DFKI); Centre for Cognitive Science; Hessian.AI",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "europe",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2023-11 (arXiv v1 2023-11-04; first tag v0.1.0 2023-12-21)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v1.1.0 released 2026-03-10: MjWarp backend, MuJoCo/MJX/MjWarp >= 3.5, models outsourced to a separate repo, new MjSpec interface",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "v1.0.1 released 2025-04-18: rewrite with MuJoCo + MJX, JAX algorithms (PPO, GAIL, AMP, DeepMimic), 12 humanoid + 4 quadruped envs, 22,000+ retargeted mocap datasets",
       "level": "verified",
       "sources": [
        "s3"
       ]
      }
     ]
    },
    "version": {
     "value": "v1.1.0 (GitHub release, 2026-03-10)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "NeurIPS 2023 6th Robot Learning Workshop (poster); arXiv v2 2023-11-30",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "locomotion"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "humanoid",
      "legged"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Read from environment module file names."
    },
    "scene": {
     "value": [],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "reward",
      "fidelity"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "v1 metrics from README/docs."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "None (: no leaderboard or results table in docs)"
    },
    "license_code": {
     "value": "MIT ('Copyright (c) 2024 Al-Hafez')",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "CC-BY-NC-ND-4.0 on the Hugging Face dataset card robfiras/loco-mujoco-datasets (default and LAFAN1 data)",
       "level": "verified",
       "sources": [
        "s9"
       ]
      },
      {
       "value": "AMASS data must be downloaded separately 'due to their licensing'; MyoSkeleton requires accepting a licence before download",
       "level": "verified",
       "sources": [
        "s5"
       ]
      }
     ]
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "display": "allowed for code; non-commercial for bundled datasets (CC-BY-NC-ND)",
     "note": "Not legal advice."
    },
    "sim_to_real": {
     "value": "claimed",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Claimed (weak): paper says users can add domain randomization, 'reducing the sim-to-real gap'; no real-robot results)",
     "note": "Weak: one paper sentence says built-in domain randomization reduces the sim-to-real gap. No real-robot experiments or paired measurements in the paper; none found elsewhere."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Release in 2026-03; repo pushed 2026-08-22."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion (full text)",
     "url": "https://arxiv.org/html/2311.02496v2",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-11"
    },
    "s2": {
     "title": "LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion",
     "url": "https://arxiv.org/abs/2311.02496",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-11"
    },
    "s3": {
     "title": "robfiras/loco-mujoco on GitHub (releases)",
     "url": "https://api.github.com/repos/robfiras/loco-mujoco/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "NeurIPS LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion",
     "url": "https://neurips.cc/virtual/2023/77280",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "robfiras/loco-mujoco on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/robfiras/loco-mujoco/master/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "robfiras/loco-mujoco on GitHub (contents)",
     "url": "https://api.github.com/repos/robfiras/loco-mujoco/contents/loco_mujoco/environments/humanoids",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "Welcome to LocoMuJoCo! — LocoMuJoCo v1.1.0 documentation",
     "url": "https://loco-mujoco.readthedocs.io/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "robfiras/loco-mujoco on GitHub (license)",
     "url": "https://api.github.com/repos/robfiras/loco-mujoco/license",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "robfiras/loco-mujoco-datasets on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/robfiras/loco-mujoco-datasets",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s10": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2311.02496",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s11": {
     "title": "robfiras/loco-mujoco on GitHub (repository)",
     "url": "https://api.github.com/repos/robfiras/loco-mujoco",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s12": {
     "title": "HumanoidBench: Simulated Humanoid Benchmark for Whole-Body Locomotion and Manipulation (full text)",
     "url": "https://arxiv.org/html/2403.10506v2",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-03"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "manipulathor",
   "name": "ManipulaTHOR",
   "full_name": "ManipulaTHOR ArmPointNav",
   "aliases": [
    "ManipulaTHOR",
    "ArmPointNav",
    "APND (Arm PointNav Dataset)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores a simulated mobile arm on navigating to, picking up and placing objects.",
   "summary": {
    "text": "Simulated task where a mobile arm must fetch an object and carry it to a target point in a kitchen.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Kiana Ehsani, Winson Han, Alvaro Herrasti, Eli VanderBilt, Luca Weihs, Eric Kolve, Aniruddha Kembhavi, Roozbeh Mottaghi; repo in allenai organisation",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2021-04 (arXiv v1 2021-04-22)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "CVPR 2021, oral (arXiv comment)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "GitHub repo archived (read-only); last push 2023-02-07; no releases",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API archived=true. Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "mobile-manipulation"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "ArmPointNav: navigate to object, pick up, carry to target (x, y, z), release. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Simulated arm 'inspired by Kinova'; text says three-jointed, Fig. 2 caption says four joints (internal conflict). Abstract grasp: object picked if it intersects a sphere at the end effector. Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "kitchen"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scale": {
     "value": "APND: 30 kitchen scenes (20/5/5), 150+ object categories (69 interactable), 12 pickupable target categories (6 seen, 6 novel), 60 tasks per object per scene, about 72k object locations",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "SRwD (success without disturbance), SR, PuSR (pick-up success), plus episode lengths. Checked 2026-10-10."
    },
    "top_score": {
     "value": "Paper baseline (depth model) Test-SeenObj: SRwD 39.4%, PuSR 89.9%, SR 68.7%; Test-NovelObj: SRwD 32.7%, SR 62.1%",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: no leaderboard found)",
     "note": "Looked in repo and AI2-THOR site. Checked 2026-10-10."
    },
    "license_code": {
     "value": "MIT (LICENSE file; includes original copyrights of Ilya Kostrikov and Facebook)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API reports NOASSERTION because of the combined copyright header. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "No real-robot use reported in the paper."
    },
    "citations": {
     "value": 169,
     "display": "169",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "status": {
     "value": "dormant",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Dormant (archived)",
     "note": "Checked 2026-10-10."
    },
    "version": {
     "value": "No releases or tags; repo archived",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "license_data": {
     "value": "APND data under the repo's MIT licence; no separate dataset licence found",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "ManipulaTHOR: A Framework for Visual Object Manipulation",
     "url": "https://arxiv.org/abs/2104.11213",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-04"
    },
    "s2": {
     "title": "allenai/manipulathor on GitHub (repository)",
     "url": "https://github.com/allenai/manipulathor",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "ManipulaTHOR: A Framework for Visual Object Manipulation (full text)",
     "url": "https://arxiv.org/html/2104.11213",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-04"
    },
    "s4": {
     "title": "allenai/manipulathor on GitHub (blob)",
     "url": "https://github.com/allenai/manipulathor/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2104.11213",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "maniskill",
   "name": "ManiSkill",
   "full_name": "ManiSkill (v1)",
   "aliases": [
    "ManiSkill",
    "SAPIEN Manipulation Skill Benchmark",
    "ManiSkill 1",
    "ManiSkill Challenge 2021",
    "ManiSkill-Legacy"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated tasks with a held-out test set and a mean-success-rate metric for robot policies; ran as a scored challenge.",
   "summary": {
    "text": "Simulated benchmark of 4 tasks on 162 articulated objects, testing whether manipulation skills transfer to unseen object instances.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "UC San Diego",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "All 9 authors at UC San Diego (Hao Su is last author). Challenge sponsored by Qualcomm (README acknowledgements)."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2021-07",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "arXiv v1 2021-07-30; README: initial version released 2021-07-29."
    },
    "latest_update": {
     "value": "2022-08: README update announcing the ManiSkill 2022 challenge (2022-08-01); last push 2022-08-24",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Push date from GitHub search API (https://api.github.com/search/repositories?q=ManiSkill+in:name+user:haosulab)."
    },
    "version": {
     "value": "Unversioned; repo renamed ManiSkill-Legacy",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Repo description: SAPIEN Manipulation Skill Benchmark (NeurIPS 2021 Track on Datasets and Benchmarks)."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "SAPIEN full-physics simulator."
    },
    "capability": {
     "value": [
      "mobile-manipulation",
      "bimanual",
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "mobile-manipulator",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Moving platform (Sciurus robot body) with one or two Franka Panda arms."
    },
    "robots": {
     "value": "Sciurus-based mobile platform with one or two Franka Panda arms",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Household articulated objects (cabinets, swivel chairs, buckets) without room scenes; mapping to 'home' is ours."
    },
    "scale": {
     "value": "4 tasks; 162 objects in 3 categories; ~36,000 trajectories; ~1.5M frames",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Sec. 2.5."
    },
    "trials": {
     "value": "100 evaluation trajectories per single environment; 50 per test environment over 10 test environments (paper experiments)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Tables 2-3 captions."
    },
    "evaluator": {
     "value": "organiser-run",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "The organisers (challenge test server) and self-reported (local evaluation script)",
     "note": "README: temporary test server launched 2022-02-16; evaluate_policy.py for local use."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Official (ManiSkill Challenge 2021); site unreachable 2026-10-10)",
     "note": "Challenge existence verified in README/paper. https://sapien.ucsd.edu/challenges/maniskill2021 refused connection (WebFetch and curl), so current content is unknown."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "README and LICENSE state no licence for assets or demonstrations. Objects are re-modelled from PartNet-Mobility, whose terms we did not check."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Paper limitations section states the authors had not yet conducted sim-to-real experiments. No later study found."
    },
    "citations": {
     "value": 275,
     "display": "275 (Semantic Scholar, 2026-10-10)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 156,
     "display": "156 (ManiSkill-Legacy, 2026-10-10)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "superseded",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Superseded (by ManiSkill2, then ManiSkill3)",
     "note": "ManiSkill3 paper: builds upon ManiSkill 1 and 2."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "ManiSkill: Generalizable Manipulation Skill Benchmark with Large-Scale Demonstrations",
     "url": "https://arxiv.org/pdf/2107.14483",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-07"
    },
    "s2": {
     "title": "ManiSkill: Generalizable Manipulation Skill Benchmark with Large-Scale Demonstrations",
     "url": "https://arxiv.org/abs/2107.14483",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-07"
    },
    "s3": {
     "title": "haosulab/ManiSkill-Legacy on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/haosulab/ManiSkill-Legacy/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "https://api.github.com/search/repositories?q=ManiSkill+in:name+user:haosulab",
     "url": "https://api.github.com/search/repositories?q=ManiSkill+in:name+user:haosulab",
     "type": "repo",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "haosulab/ManiSkill-Legacy on GitHub (main)",
     "url": "https://raw.githubusercontent.com/haosulab/ManiSkill-Legacy/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s7": {
     "title": "ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI",
     "url": "https://arxiv.org/pdf/2410.00425",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "maniskill2",
   "name": "ManiSkill2",
   "aliases": [
    "ManiSkill2: A Unified Benchmark for Generalizable Manipulation Skills",
    "ManiSkill 2",
    "mani_skill2",
    "ManiSkill2 Challenge 2022"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated task families with success metrics, held-out objects and a public challenge evaluation for robot policies.",
   "summary": {
    "text": "Simulated benchmark of 20 manipulation task families, rigid and soft body, with 2,000+ objects and 4M+ demo frames.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "UC San Diego; Tsinghua University",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "15 authors; Hao Su last author."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Lead institution UC San Diego."
    },
    "first_release": {
     "value": "2022-08",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "First release tag v0.2.0 on 2022-08-15; arXiv v1 2023-02-09."
    },
    "latest_update": {
     "value": "v0.5.3 released 2023-09-22 (RNG seeding bug fix)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "ManiSkill3 README: original ManiSkill2 code lives at the v0.5.3 tag."
    },
    "version": {
     "value": "0.5.3",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "SAPIEN rigid-body simulation plus custom GPU MPM (Warp) for soft bodies."
    },
    "capability": {
     "value": [
      "manipulation",
      "mobile-manipulation",
      "bimanual"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "mobile-manipulator",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "'stationary/mobile-base, single/dual-arm'."
    },
    "robots": {
     "value": "panda, mobile_panda, xmate3 robot definitions",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Files in mani_skill2/agents/robots at v0.5.3."
    },
    "scene": {
     "value": [
      "tabletop",
      "home"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Tabletop rigid/soft tasks plus household articulated-object tasks; mapping is ours."
    },
    "scale": {
     "value": "20 task families; 2,000+ object models; 4M+ demonstration frames",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Same numbers on project site https://maniskill2.github.io/."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Appendix: e.g. PickCube success = cube within 2.5 cm of goal and robot static."
    },
    "trials": {
     "value": "100 episodes per task with varied initial states (most rigid tasks)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Appendix task protocols."
    },
    "evaluator": {
     "value": "organiser-run",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "The organisers (cloud-based challenge evaluation) and self-reported)",
     "note": "'we implement a cloud-based evaluation system'."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Official (ManiSkill2 Challenge 2022); site unreachable 2026-10-10)",
     "note": "https://sapien.ucsd.edu/challenges/maniskill/2022/ refused connection; project site invites to the challenge without results."
    },
    "license_code": {
     "value": "Apache-2.0 for rigid-body environments; soft-body environments follow NVIDIA Source Code License for Warp",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "README 'License' section at tag v0.5.3; repo LICENSE file is Apache-2.0."
    },
    "license_data": {
     "value": "Assets: CC BY-NC 4.0 (README). Demonstrations: Hugging Face card haosulab/ManiSkill2 says apache-2.0",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Two primary sources give different licences for different artefacts; record both."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Code permissive, assets non-commercial, soft-body code under Warp licence. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "One point-cloud RL policy for PickCube: 91.0% success in simulation, 60.0% over 50 real trials (ROKAE xMate3Pro arm, Robotiq 2F-140, RealSense D415). Pinch: same motion-planned action sequence run in sim and real, compared qualitatively. Measured by the ManiSkill2 authors; no correlation statistic."
    },
    "citations": {
     "value": 451,
     "display": "451 (Semantic Scholar, 2026-10-10)",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "superseded",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Superseded (by ManiSkill3)"
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "ManiSkill2: A Unified Benchmark for Generalizable Manipulation Skills",
     "url": "https://arxiv.org/pdf/2302.04659",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-02"
    },
    "s2": {
     "title": "mani-skill/ManiSkill on GitHub (releases)",
     "url": "https://api.github.com/repos/mani-skill/ManiSkill/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "mani-skill/ManiSkill on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/mani-skill/ManiSkill/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "ManiSkill2: A Unified Benchmark for Generalizable Manipulation Skills",
     "url": "https://arxiv.org/abs/2302.04659",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2023-02"
    },
    "s5": {
     "title": "mani-skill/ManiSkill on GitHub (file robots?ref=v0.5.3)",
     "url": "https://api.github.com/repos/mani-skill/ManiSkill/contents/mani_skill2/agents/robots?ref=v0.5.3",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "mani-skill/ManiSkill on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/mani-skill/ManiSkill/v0.5.3/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "haosulab/ManiSkill2 on Hugging Face (dataset)",
     "url": "https://huggingface.co/api/datasets/haosulab/ManiSkill2",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "maniskill3",
   "name": "ManiSkill3",
   "full_name": "ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI",
   "aliases": [
    "ManiSkill 3",
    "mani_skill>=3.0.0",
    "ManiSkill",
    "Demonstrating GPU Parallelized Robot Simulation and Rendering for Generalizable Embodied AI with ManiSkill3"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "A simulator that ships scored robot tasks (success conditions, rewards, demonstrations) and real-to-sim evaluation twins; papers report success rates on its tasks. Simulators that host tasks are in the taxonomy's kind vocabulary.",
   "summary": {
    "text": "ManiSkill3 is an open-source robot simulator from UC San Diego's Hao Su lab that runs many copies of a task in parallel on one GPU. Its documentation lists 51 tasks, from tabletop picking to humanoids and four real-to-sim test scenes, and it has no single benchmark score.",
    "short": "ManiSkill3 is an open-source robot simulator that runs many copies of a task in parallel on one GPU. Its documentation lists 51 tasks, and it has no single benchmark score.",
    "sources": [
     "s2",
     "s13",
     "s9"
    ]
   },
   "facts": {
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The paper's conclusion calls ManiSkill3 a 'framework/benchmark'; the GitHub description says 'an open source GPU parallelized robotics simulator and benchmark'. We file it as a simulator that hosts many tasks because it has no single fixed task set or score.",
     "short": "A simulator with a library of tasks"
    },
    "kind_secondary": {
     "value": [
      "benchmark"
     ],
     "display": "Also a task collection with success conditions, demonstrations and tuned baselines",
     "level": "verified",
     "sources": [
      "s2",
      "s13",
      "s15"
     ],
     "checked": "2026-10-10"
    },
    "lineage": {
     "value": "third generation of ManiSkill",
     "display": "ManiSkill (2021) and ManiSkill2 (2023) were fixed benchmarks with challenges. ManiSkill3 (2024) rebuilds the code for GPU-parallel simulation and rendering. ManiSkill2 and ManiSkill3 share one GitHub repository; ManiSkill2 code stays at tag v0.5.3.",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s48",
      "s49"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ManiSkill (v1), 2021",
       "display": "arXiv 2021-07-30; NeurIPS 2021 Datasets and Benchmarks. 4 tasks (OpenCabinetDoor, OpenCabinetDrawer, PushChair, MoveBucket) on 162 objects; about 36,000 demonstrations. Code in haosulab/ManiSkill-Legacy, last push 2022-08-24.",
       "level": "verified",
       "sources": [
        "s48",
        "s50"
       ]
      },
      {
       "value": "ManiSkill2, 2023",
       "display": "arXiv 2023-02-09; ICLR 2023. 20 task families, 2,000+ object models, 4M+ demonstration frames, rigid and soft body. CPU simulation with a shared render server. Last release mani-skill2 0.5.3 on 2023-09-22.",
       "level": "verified",
       "sources": [
        "s49",
        "s51",
        "s7",
        "s9",
        "s2"
       ]
      },
      {
       "value": "Bridge releases 0.6.0.dev0 to dev3",
       "display": "mani-skill2 0.6.0 development releases (2023-12-14 to 2024-01-12) integrated SAPIEN 3 and scene loading, before the package was renamed mani-skill 3.0.",
       "level": "verified",
       "sources": [
        "s51",
        "s7"
       ]
      },
      {
       "value": "What changed in ManiSkill3",
       "display": "The paper says GPU-parallelized simulation and rendering is what distinguishes it from its predecessors. Task IDs end in -v1. ManiSkill1's cabinet-opening tasks reappear as OpenCabinetDrawer-v1 and OpenCabinetDoor-v1. ManiSkill2's soft-body tasks are not ported; the docs say they remain in the ManiSkill2 codebase.",
       "level": "verified",
       "sources": [
        "s2",
        "s13",
        "s50"
       ]
      },
      {
       "value": "Citation rule",
       "display": "README: cite the ManiSkill3 paper for mani_skill>=3.0.0 and the ManiSkill2 paper for version 0.5.3 or lower.",
       "level": "verified",
       "sources": [
        "s9"
       ]
      }
     ],
     "short": "Successor to ManiSkill (2021) and ManiSkill2 (2023)"
    },
    "version": {
     "value": "3.0.1",
     "display": "3.0.1, released 2026-04-21, is the first release without the 'beta' label. Betas ran from 3.0.0.b2 (2024-05-02) to 3.0.0b22 (2025-12-05).",
     "level": "verified",
     "sources": [
      "s7",
      "s8",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "There is no 3.0.0 release: PyPI goes from 3.0.0b22 to 3.0.1. A tag v3.0.0b23 exists without a release. The roadmap plans to replace the PhysX backend with Newton/MuJoCo Warp, which would change the physics behind task scores.",
     "items": [
      {
       "value": "3.0.0b11 (2024-10)",
       "display": "Release notes for b12 say an RNG bug in b11 sometimes stopped environments from randomizing on reset, and ask users to upgrade.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "3.0.0b10 (2024-10-01)",
       "display": "Added the four SIMPLER Bridge real-to-sim twins and aligned evaluation metrics and baseline reporting.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "Planned backend change",
       "display": "Roadmap: switch the PhysX backend to Newton/MuJoCo Warp.",
       "level": "verified",
       "sources": [
        "s46"
       ]
      }
     ],
     "short": "3.0.1, from April 2026"
    },
    "publishers": {
     "value": [
      "UC San Diego",
      "Hillbot",
      "Carnegie Mellon University",
      "TU Dresden",
      "Tsinghua University",
      "King's College London"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "23 authors; Hao Su is last author. 19 list UC San Diego, and 7 of those also list Hillbot. Hillbot and the Qualcomm Embodied AI Fund are thanked for support.",
     "items": [
      {
       "value": "UC San Diego",
       "display": "Hao Su lab; 19 of 23 authors",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Hillbot",
       "display": "Robotics company; second affiliation of 7 authors including Stone Tao, Fanbo Xiang and Hao Su",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Carnegie Mellon University",
       "display": "Chen Bao",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "TU Dresden",
       "display": "Roberto Calandra",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Tsinghua University",
       "display": "Rui Chen",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "King's College London",
       "display": "Shan Luo",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "University lab lead; several authors also work at Hillbot, a company."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Lead organisation is UC San Diego."
    },
    "first_release": {
     "value": "2024-03",
     "display": "First 3.x package on PyPI on 2024-03-09 (3.0.0.dev0); first GitHub release 2024-03-28; arXiv v1 2024-10-01; RSS 2025.",
     "level": "verified",
     "sources": [
      "s8",
      "s7",
      "s1",
      "s3",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "arXiv v2 (2025-05-30) is the RSS 2025 demonstration-track text with a new title. The paper was also an oral at the ICLR 2025 Robot Learning Workshop.",
     "short": "March 2024. The paper followed in October 2024 and was published at RSS 2025."
    },
    "latest_update": {
     "value": "2026-08",
     "display": "Last commit on main 2026-08-02. Last release 3.0.1 on 2026-04-21.",
     "level": "verified",
     "sources": [
      "s35",
      "s34",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "GitHub reports the last push to any branch on 2026-08-04.",
     "short": "August 2026, with new commits"
    },
    "capability": {
     "value": [
      "manipulation",
      "mobile-manipulation",
      "dexterous",
      "bimanual",
      "locomotion"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "The paper lists 12 task categories: table-top, mobile manipulation, room-scale scenes, quadruped/humanoid locomotion, humanoid/bi-manual, multi-agent, drawing/cleaning, dexterous, vision-tactile, classic control, digital twins and soft body. The taxonomy has no value for drawing, vision-tactile or classic control."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "Start states are randomized every episode. Some tasks also change object geometry (for example peg shapes in PegInsertionSide). No held-out object split is documented.",
     "level": "inferred",
     "sources": [
      "s13",
      "s16",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The docs' evaluation setup re-randomizes every episode (reconfiguration_freq=1), which 'randomizes object geometries if the task has object randomization'. We found no statement that evaluation objects are unseen in training.",
     "short": "Start positions vary, and some tasks also vary object shapes."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "SAPIEN 3 with PhysX (GPU and CPU backends)",
     "level": "verified",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Rendering uses SAPIEN's parallel rasterizer; ray tracing is available without parallelization. GPU simulation is supported on Linux with NVIDIA GPUs only (README system table).",
     "short": "SAPIEN 3 with PhysX GPU simulation"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "mobile-manipulator",
      "humanoid",
      "legged",
      "dexterous-hand",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s13",
      "s54"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "35 robot configurations in the docs",
     "display": "Docs have 35 robot pages (our count), including Franka Panda, xArm 6/7, UR10e, WidowX 250S, WidowX AI, SO100, Koch v1.1, Fetch, Google Robot, Unitree G1, H1 and Go2, ANYmal C, Allegro, D'Claw, Inspire hands and TriFinger. The paper says 20+ robots.",
     "level": "inferred",
     "sources": [
      "s54",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Counted from docs/source/robots/*/index.md on 2026-10-10. Some pages are variants of one robot (wrist camera, simplified legs).",
     "short": "35 robot setups in the docs"
    },
    "scene": {
     "value": [
      "tabletop",
      "home"
     ],
     "display": "Most scored tasks are tabletop. ManiSkill-HAB adds apartment rearrangement in ReplicaCAD scenes.",
     "level": "inferred",
     "sources": [
      "s13",
      "s23",
      "s44"
     ],
     "checked": "2026-10-10",
     "note": "RoboCasa kitchens, ReplicaCAD and AI2-THOR scenes load in ManiSkill3, but the docs say none of these large scene datasets has trainable tasks with success/fail conditions yet; RoboCasaKitchen-v1 has no success condition. So 'kitchen' is not tagged."
    },
    "tasks": {
     "value": 51,
     "display": "51 task entries in 9 documentation categories (our count). The paper groups tasks into 12 categories.",
     "level": "inferred",
     "sources": [
      "s13",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Counted 2026-10-10 from the generated task tables and pages: table-top 21, control 9, humanoid 4, digital twins 4, quadruped 3, mobile manipulation 3, dexterous 3 task families (some with difficulty levels 0 to 4), drawing 3, external 1 (ManiSkill-HAB). Of the 44 entries in task tables, 35 have success/fail conditions and 17 have demonstrations.",
     "short": "51 listed tasks"
    },
    "scenes": {
     "value": 3,
     "display": "3 room-scale scene datasets: ReplicaCAD, AI2-THOR-family scenes (for example ArchitecTHOR) and RoboCasa kitchens",
     "level": "verified",
     "sources": [
      "s23",
      "s24"
     ],
     "checked": "2026-10-10",
     "note": "Used for exploration and data generation. Only ManiSkill-HAB defines scored tasks in ReplicaCAD.",
     "short": "3 scene datasets"
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No object count in the paper, README or docs. Tasks load objects from YCB, PartNet-Mobility cabinets, OakInk-v2 and others through the asset registry; we did not count them."
    },
    "demonstrations": {
     "value": 16,
     "display": "Official demonstrations for 16 tasks on Hugging Face, about 0.47 GB of compressed files. Motion-planning sets hold 1,000 trajectories per task (two tasks checked). The paper says 'millions of demonstration frames'.",
     "level": "inferred",
     "sources": [
      "s20",
      "s55",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Task count and size summed by us from the Hugging Face file listing (16 zip files, 473,816,098 bytes). Episode counts read from trajectory.json for PickCube-v1 and PegInsertionSide-v1. Files store states and actions only; observations are regenerated with the replay tool.",
     "items": [
      {
       "value": "Sources",
       "display": "Motion planning, RL policies and teleoperation, labelled in each file's metadata (for example 10 teleoperated PickCube-v1 demos)",
       "level": "verified",
       "sources": [
        "s20",
        "s55"
       ]
      },
      {
       "value": "Per-backend files",
       "display": "File names record the controller and the simulation backend (physx_cpu or physx_cuda)",
       "level": "verified",
       "sources": [
        "s17",
        "s20"
       ]
      }
     ],
     "short": "Demonstrations for 16 tasks"
    },
    "scoring": {
     "value": [
      "success-rate",
      "reward"
     ],
     "level": "verified",
     "sources": [
      "s16",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Tasks define success/fail conditions and, for most, dense rewards for RL. There is no benchmark-wide headline score."
    },
    "metric_detail": {
     "value": "success_once and success_at_end",
     "display": "Each episode records success_once (succeeded at any step), success_at_end (succeeded at the last step), the matching fail flags and the return. The docs say learning-from-demonstration work typically reports success_once. No benchmark-wide average.",
     "level": "verified",
     "sources": [
      "s16",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "The two success metrics can differ for the same rollouts, so a paper must say which one it reports.",
     "short": "Success for each task, with two success metrics"
    },
    "trials": {
     "value": "not fixed",
     "display": "The docs give an evaluation script (no early resets, re-randomize every episode, 64 parallel environments in the example) but no required number of episodes.",
     "level": "verified",
     "sources": [
      "s16",
      "s17",
      "s2",
      "s39"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ManiSkill3 paper, imitation baselines",
       "display": "Best success rate obtained during training (Tables II and III); PerAct evaluated over 100 episodes",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "RDT-1B five-task protocol",
       "display": "250 trials per task (10 seeds x 25), mean and standard deviation",
       "level": "verified",
       "sources": [
        "s39"
       ]
      },
      {
       "value": "RL baselines",
       "display": "Evaluation curves shared as Weights & Biases reports; RL standard benchmark small set listed in the docs",
       "level": "verified",
       "sources": [
        "s15"
       ]
      }
     ],
     "short": "Not fixed. It varies by paper."
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s39",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "The paper's RL curves show 95% confidence intervals over 5 seeds and the Koch sim2real curves show 95% intervals; its imitation tables give single numbers. RDT reports mean and standard deviation over 10 seeds; E0 gives single numbers."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s9",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions. Each paper runs its own evaluation."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s9",
      "s15",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard in the README, docs or paper. The authors share baseline training runs as Weights & Biases reports, which are not a results table for outside methods."
    },
    "top_score": {
     "value": "no headline score",
     "display": "ManiSkill3 has no benchmark-wide score. The most reused subset we found is RDT's five-task protocol (PegInsertionSide, PickCube, StackCube, PlugCharger, PushCube), where the best average we found is 55.2% (E0, November 2025).",
     "level": "inferred",
     "sources": [
      "s39",
      "s40",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Results across papers use different demonstrations, controllers and checkpoints, so we do not chart them.",
     "items": [
      {
       "value": 53.6,
       "display": "RDT-1B, five-task average 53.6% (13.2 / 77.2 / 74.0 / 1.2 / 100); Diffusion Policy 30.2%, OpenVLA 4.8%, Octo 0.0% in the same table",
       "level": "verified",
       "sources": [
        "s39"
       ],
       "note": "5,000 motion-planning trajectories; 250 trials per task; released 2024-12-17 per the README news."
      },
      {
       "value": 55.2,
       "display": "E0, five-task average 55.2% (24.0 / 76.0 / 72.0 / 4.0 / 100); pi0 42.4%, pi0.5 43.2%, pi0-FAST 44.8% in the same table",
       "level": "verified",
       "sources": [
        "s40"
       ],
       "note": "E0 reprints RDT's baseline rows. Its OpenVLA StackCube cell reads 80.0 while RDT reports 8%; the row average (4.8) matches 8%."
      },
      {
       "value": "PickCube, StackCube with Diffusion Policy",
       "display": "ManiSkill3 paper: Diffusion Policy from RGB with 100 demos scores 0.76 on PickCube and 0.61 on StackCube; RDT's Diffusion Policy row scores 40.0% and 80.0%",
       "level": "verified",
       "sources": [
        "s2",
        "s39"
       ],
       "note": "Different data (100 vs 5,000 demos), controllers and checkpoint rules."
      }
     ],
     "short": "No headline score"
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s10",
      "s9",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE is Apache-2.0. Since 2025-09-14 a LICENSE-3RD-PARTY file adds the BSD licence of copied PyTorch3D code, after issue #1208 flagged a file under LGPL-3.0. The README says rigid-body environments use 'fully permissive licenses (e.g., Apache-2.0)'."
    },
    "license_data": {
     "value": "Apache-2.0",
     "display": "Demonstration dataset card on Hugging Face: apache-2.0",
     "level": "verified",
     "sources": [
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "The card covers haosulab/ManiSkill_Demonstrations. The README's asset licence (CC BY-NC 4.0) may also apply to rendered observations of those assets; the card does not say."
    },
    "license_assets": {
     "value": "CC-BY-NC-4.0",
     "display": "README: 'The assets are licensed under CC BY-NC 4.0'. Individual sources carry other labels (items).",
     "level": "verified",
     "sources": [
      "s9",
      "s25",
      "s26",
      "s24"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT between the blanket README statement and per-source labels. Terms for PartNet-Mobility assets (downloaded from UCSD storage) could not be checked: the SAPIEN site did not respond on 2026-10-10.",
     "items": [
      {
       "value": "Robot models",
       "display": "Own LICENSE files: Allegro (BSD-style, SimLab), D'Claw (Apache-2.0, ROBEL), Unitree G1 (BSD-3-Clause), Koch (Apache-2.0, Hugging Face), SO100 (Apache-2.0)",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Scene copies on Hugging Face",
       "display": "haosulab/ReplicaCADRearrange and haosulab/AI2THOR cards say cc-by-4.0; haosulab/RoboCasa says mit, while RoboCasa's own README says CC BY 4.0",
       "level": "verified",
       "sources": [
        "s26"
       ]
      }
     ],
     "short": "CC BY-NC 4.0 according to the README. Other labels conflict with it."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s9",
      "s8",
      "s20",
      "s19",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "pip install mani_skill; assets and demonstrations download from public Hugging Face and UCSD links without registration. Since a 2026-06-07 link update, the docs point to mani-skill/ManiSkill_Demonstrations, which returns HTTP 401; the download tool still uses the working haosulab repository.",
     "short": "Open. The package installs with pip, and the data is on Hugging Face."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s9",
      "s10",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "Code (Apache-2.0) allows commercial use. The README licenses the bundled assets under CC BY-NC 4.0, which bars commercial use, even though some individual assets carry permissive labels. A company could use the simulator with its own assets. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Measured by the authors on the 4 Bridge real-to-sim twins: Pearson r 0.9284, MMRV 0.0147 over 3 policies. A sim-to-real cube task was compared by plot only. No independent replication.",
     "level": "inferred",
     "sources": [
      "s2",
      "s29",
      "s27",
      "s36",
      "s37",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; the numbers are verified. (1) Real-to-sim: Octo-Base, Octo-Small and RT-1-X on ManiSkill3's GPU port of four SIMPLER WidowX tasks (spoon on towel, carrot on plate, stack cube, eggplant in basket). Figure 25's legend has 3 policies and 8 task-metric pairs (4 tasks, each scored for grasp and for success), all pooled into one r. Its 'real' values equal SIMPLER's published real results (Table V), so no new real runs were needed or reported (our inference from matching values). The paper does not say whether r was pooled over points or averaged per task as SIMPLER does (SIMPLER reports r 0.890, MMRV 0.014 for Bridge). The ManiSkill3 and SIMPLER author lists overlap. (2) Sim-to-real (added in v2, 2025-05): an RL policy trained in simulation picked cubes with a $300 Koch arm, 22/24 = 91.6% real success over 3 runs; Figure 13 plots sim and real success for 14 checkpoints per run, called 'a good correlation' with no statistic. (3) Related but not ManiSkill3 tasks: SureSim (2025-10) built its own ManiSkill3 twin of a Franka setup and reports a paired sim-real correlation of 0.72 for a diffusion policy; Squint (2026-02) reports sim and real success side by side for 5 methods on 8 custom SO-101 tasks, with no statistic. GPUSimBench (2026-07) compared ManiSkill 3.0.0b22 physics to a real inclined-collision apparatus (lowest distance of 7 simulators, 2.520 cm) without policies. The vision-tactile peg result (95.08% real) is cited from earlier work.",
     "short": "The authors checked 4 digital twins and found r = 0.93."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation."
    },
    "derived_benchmarks": {
     "value": [
      "ManiSkill-HAB",
      "SimplerEnv ManiSkill3 twins",
      "RL4VLA task set",
      "MIKASA-Robo-VLA",
      "RoboFactory"
     ],
     "display": "Benchmarks and task sets built on ManiSkill3",
     "level": "verified",
     "sources": [
      "s44",
      "s14",
      "s41",
      "s45"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ManiSkill-HAB",
       "display": "2024-12. Home rearrangement (TidyHouse, PrepareGroceries, SetTable) in ReplicaCAD apartments; listed in the docs as an external task",
       "level": "verified",
       "sources": [
        "s44",
        "s13"
       ]
      },
      {
       "value": "SimplerEnv ManiSkill3 twins",
       "display": "4 Bridge WidowX evaluation twins ported to GPU inside ManiSkill3 (since 3.0.0b10); inference code in SimplerEnv's maniskill3 branch",
       "level": "verified",
       "sources": [
        "s14",
        "s33",
        "s28"
       ]
      },
      {
       "value": "RL4VLA task set",
       "display": "2025-05. WidowX pick-and-place tasks with 4,352 object, receptacle and scene combinations, reused by piRL and RLinf-VLA",
       "level": "verified",
       "sources": [
        "s41",
        "s43",
        "s42"
       ]
      },
      {
       "value": "Docs community list",
       "display": "MIKASA-Robo-VLA, RoboFactory, Molmospaces and GSWorld are listed as community projects built on ManiSkill",
       "level": "verified",
       "sources": [
        "s45"
       ],
       "note": "Listed by the ManiSkill docs; we did not open each project."
      }
     ],
     "note": "Not a complete list.",
     "short": "Several task sets built on it"
    },
    "citations": {
     "value": 355,
     "display": "355 (Semantic Scholar; 37 influential)",
     "level": "verified",
     "sources": [
      "s47",
      "s57"
     ],
     "checked": "2026-10-10",
     "note": "For comparison on the same day: ManiSkill (v1) 275, ManiSkill2 451.",
     "short": "355"
    },
    "github_stars": {
     "value": 3389,
     "display": "3,389 stars, 552 forks (mani-skill/ManiSkill)",
     "level": "verified",
     "sources": [
      "s6",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The repository was created 2022-08-02 for ManiSkill2, so stars include ManiSkill2 users. It moved from the haosulab to the mani-skill GitHub organisation in 2026-06.",
     "short": "3,389"
    },
    "dataset_downloads": {
     "value": 3726,
     "display": "3,726 (Hub 'downloads' field), 31,338 all time, 10 likes: haosulab/ManiSkill_Demonstrations",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API. We did not check the time window behind the 'downloads' field.",
     "short": "3,726 on Hugging Face"
    },
    "used_by": {
     "value": "No count of papers reporting ManiSkill3 results was found. Named users below.",
     "level": "verified",
     "sources": [
      "s39",
      "s41",
      "s42",
      "s43",
      "s40",
      "s37",
      "s38",
      "s36"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RDT-1B",
       "display": "Tsinghua, 2024-12: five-task ManiSkill simulation benchmark in its README; the ManiSkill docs list RDT-1B, Octo and RT-X as VLAs tested through ManiSkill",
       "level": "verified",
       "sources": [
        "s39",
        "s56"
       ]
      },
      {
       "value": "RL4VLA",
       "display": "2025-05: RL fine-tuning of VLAs on WidowX tasks built in ManiSkill",
       "level": "verified",
       "sources": [
        "s41"
       ]
      },
      {
       "value": "RLinf-VLA",
       "display": "2025-10: RL framework; ManiSkill is one of its simulators",
       "level": "verified",
       "sources": [
        "s42"
       ]
      },
      {
       "value": "piRL",
       "display": "2025-10: ManiSkill benchmark with 4,352 pick-and-place combinations, following RL4VLA",
       "level": "verified",
       "sources": [
        "s43"
       ]
      },
      {
       "value": "E0",
       "display": "2025-11: reports the RDT five-task protocol",
       "level": "verified",
       "sources": [
        "s40"
       ]
      },
      {
       "value": "SureSim, Squint, GPUSimBench",
       "display": "2025-10 to 2026-07: evaluation-method and sim-to-real studies that run in ManiSkill3",
       "level": "verified",
       "sources": [
        "s37",
        "s38",
        "s36"
       ]
      }
     ],
     "short": "Used in vision-language-action (VLA) and reinforcement learning (RL) papers"
    },
    "industry_use": {
     "value": [
      "Hillbot",
      "Trossen Robotics",
      "Infinigence AI"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s52",
      "s53",
      "s42"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Hillbot",
       "display": "Co-developer: 7 authors list Hillbot; the paper thanks Hillbot for support",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Trossen Robotics",
       "display": "Hosts the WidowX AI robot assets for ManiSkill (repository created 2025-06-03); support merged in 3.0.0b22",
       "level": "verified",
       "sources": [
        "s52",
        "s53"
       ]
      },
      {
       "value": "Infinigence AI",
       "display": "Co-authors of RLinf-VLA, which trains VLAs with RL in ManiSkill",
       "level": "verified",
       "sources": [
        "s42"
       ]
      }
     ]
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s35",
      "s34",
      "s46"
     ],
     "checked": "2026-10-10",
     "note": "Commits on main through 2026-08-02 and a release on 2026-04-21, both within six months. The roadmap plans API clean-up and a physics backend switch.",
     "short": "Active. The latest release was in April 2026."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "There is no fixed protocol or headline score",
     "text": "ManiSkill3 is a task library, and papers pick their own tasks, demonstrations, controllers and episode counts. The docs' RL 'standard benchmark' says it has 'a small set of 8 tasks' but lists 7, one under an ID that is not registered (HumanoidPlaceAppleInBowl-v1; the registered task is UnitreeG1PlaceAppleInBowl-v1); the 50-task large set 'is still being developed'. The learning-from-demonstrations baselines page is marked work in progress. Each episode records two success metrics, success_once and success_at_end, and the docs say demonstration-learning work typically reports success_once.",
     "short": "Papers choose their own tasks, data and success metric.",
     "level": "verified",
     "sources": [
      "s15",
      "s16",
      "s17",
      "s18",
      "s58"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same method gets very different numbers",
     "text": "Diffusion Policy on PickCube: 0.76 (ManiSkill3 paper, RGB, 100 demos, best during training) and 1.00 with 1,000 demos, but 40.0% in RDT's five-task table (5,000 motion-planning demos, joint position control, checkpoint picked by validation loss). On StackCube the same rows read 0.61 and 80.0%. E0's reprint of RDT's table changes OpenVLA's StackCube cell from 8% to 80.0 while keeping the 4.8% average.",
     "short": "Depending on the paper, Diffusion Policy scores 0.76 or 0.40 on PickCube.",
     "level": "verified",
     "sources": [
      "s2",
      "s39",
      "s40"
     ],
     "status": "open"
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "CPU and GPU simulation can give different results",
     "text": "A user reported a Diffusion Policy at 70% success on the CPU backend and 12% on the GPU backend over 100 episodes of a custom task, and official motion planners failing on GPU for StackCube, PegInsertionSide, PlaceSphere and LiftPegUpright (version 3.0.0b21). The maintainer replied that the planners were not fully tested on GPU and suggested collecting and testing on CPU; the maintainer closed the issue on 2026-03-14 without a code fix. The docs advise evaluating on the same backend the demonstrations were collected on. An independent test (GPUSimBench, ManiSkill 3.0.0b22) found that parallel GPU environments started from identical states end in measurably different places (mean pairwise distance 4.76 cm), while whole runs repeat exactly.",
     "short": "One policy scored 70% in CPU simulation and 12% in GPU simulation.",
     "level": "verified",
     "sources": [
      "s30",
      "s17",
      "s36"
     ],
     "status": "open"
    },
    {
     "id": "i4",
     "type": "protocol-variance",
     "title": "Scores depend on the package version",
     "text": "Beta releases changed evaluation-relevant code: 3.0.0b11 sometimes did not randomize environments on reset (fixed in b12, 2024-10-29); b10 aligned evaluation metrics and baseline reporting; a 2026-06-23 fix corrected seeding that shrank the episode seed batch after an unseeded reset. The SimplerEnv port notes its evaluation 'is not deterministic and results may vary between runs', and the original SIMPLER results need the older main branch.",
     "short": "Beta versions changed the random-seed and evaluation code. Results therefore depend on the version.",
     "level": "verified",
     "sources": [
      "s32",
      "s33",
      "s31",
      "s28"
     ],
     "status": "open"
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Asset licences are non-commercial and inconsistent",
     "text": "The README puts all assets under CC BY-NC 4.0. Robot folders carry their own BSD or Apache-2.0 licences, and Hugging Face copies of scene datasets are labelled CC BY 4.0 or MIT; the MIT label on the RoboCasa copy differs from RoboCasa's own CC BY 4.0. A file under LGPL-3.0 sat in the Apache-2.0 code until 2025-09-14.",
     "short": "The README puts all assets under CC BY-NC 4.0. The labels on individual assets differ.",
     "level": "verified",
     "sources": [
      "s9",
      "s25",
      "s26",
      "s12",
      "s11"
     ],
     "status": "open",
     "note": "The LGPL item was addressed by pull request #1269. Not legal advice."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "There is no single ManiSkill3 score. A number from ManiSkill3 only means something next to the task, demonstration source, controller, success metric, backend and version used. Compare results only within one paper or under a shared protocol such as RDT's five tasks.",
     "short": "Compare ManiSkill3 numbers only within one paper or protocol.",
     "basis": [
      "facts.top_score",
      "facts.metric_detail",
      "issues.i1",
      "issues.i2",
      "issues.i4"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r2",
     "text": "The real-world evidence is narrow and comes from the authors. It covers a re-implementation of four SIMPLER twins (simulated copies of real test setups), checked against SIMPLER's existing real data for three policies, and one cube task on a low-cost arm. It does not show that scores on the other ManiSkill3 tasks predict real performance.",
     "short": "The real-world checks cover four digital twins and one cube task.",
     "basis": [
      "facts.sim_to_real"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r3",
     "text": "ManiSkill3's main value is speed. Many copies of a task run in parallel on one GPU, which suits reinforcement learning and large data generation. Its tasks are better read as a development testbed than as a ranking of general skill.",
     "short": "It is best used as a fast testbed during development. It does not rank general skill.",
     "basis": [
      "facts.kind",
      "facts.simulator",
      "facts.used_by"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r4",
     "text": "Teams planning commercial work should check asset licences task by task. The code is permissive, but the README puts the assets under a non-commercial licence.",
     "short": "Check asset licences before commercial use.",
     "basis": [
      "facts.license_assets",
      "facts.commercial_use",
      "issues.i5"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10"
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy (the robot's control model) will do on a real robot.",
     "sub": "The authors checked this on 4 digital twins (simulated copies of real test setups) and 1 cube task.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How methods compare on one shared score.",
     "sub": "ManiSkill3 has no headline score and no fixed test protocol.",
     "basis": [
      "facts.top_score",
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "Whether results hold across simulator settings.",
     "sub": "Switching between CPU and GPU simulation, or changing the package version, can change scores.",
     "basis": [
      "issues.i3",
      "issues.i4"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "ManiSkill3 port of SIMPLER Bridge twins",
     "date": "2024-10",
     "by": "authors",
     "method": "Octo-Base, Octo-Small and RT-1-X were run on ManiSkill3's GPU port of four SIMPLER WidowX tasks and compared with their real success rates on the same tasks, which equal SIMPLER's published real results.",
     "result": "Pearson r = 0.9284. MMRV (a measure of how often two rankings disagree) is 0.0147.",
     "authors_view": "close to the original values reported in SIMPLER",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s2",
      "s29",
      "s27"
     ]
    },
    {
     "id": "v2",
     "name": "Koch cube-picking sim2real",
     "date": "2025-05",
     "by": "authors",
     "method": "14 checkpoints from each of 3 RL training runs were evaluated 8 times each in simulation and on a real Koch arm on the same cube-picking task.",
     "result": "No statistic was reported. Success curves in simulation and on the real robot are plotted in Figure 13.",
     "authors_view": "good correlation",
     "n_policies": 42,
     "level": "verified",
     "sources": [
      "s2"
     ]
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "ManiSkill3 paper v1 and v2 full text (Sections III-E, IV-D, Appendix VII-K, VIII, Figures 13 and 25); docs digital-twin and sim2real pages; SimplerEnv maniskill3 README; SIMPLER paper Table V; SureSim (2510.04354; custom twin); Squint (2602.21203; custom tasks); GPUSimBench (2607.13059; physics only); 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (2606.10366; cites ManiSkill in related work only); PolaRiS (2512.16881), Betting for Sim-to-Real (2604.24018), Robot Policy Evaluation for Sim-to-Real Transfer (2508.11117), Active Real-World Factor-Based Evaluation (2607.14439), Beyond Binary Success (2603.13616): no ManiSkill3 mention in full text; the 2026 audit (2606.04233) does not cover ManiSkill3; web searches for ManiSkill3 sim-vs-real correlation and critiques.",
     "date": "2026-10-10"
    },
    {
     "for": "objects",
     "where": "Paper, README, docs task pages and asset registry: no object count.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (PartNet-Mobility terms)",
     "where": "sapien.ucsd.edu did not respond on 2026-10-10.",
     "date": "2026-10-10"
    },
    {
     "for": "used_by (count)",
     "where": "No tracker of papers reporting ManiSkill3 results found; the 2026 audit tracks five other benchmarks only.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "README, docs (tasks, baselines, community projects), paper.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI (arXiv abstract page, v1 and v2 history)",
     "url": "https://arxiv.org/abs/2410.00425",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "ManiSkill3 paper, full text v2 (retitled 'Demonstrating GPU Parallelized Robot Simulation and Rendering for Generalizable Embodied AI with ManiSkill3')",
     "url": "https://arxiv.org/pdf/2410.00425v2",
     "type": "paper",
     "publisher": "arXiv (RSS 2025 text)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "ManiSkill3 paper, full text v1",
     "url": "https://arxiv.org/pdf/2410.00425v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Robotics: Science and Systems XXI, paper 21 (proceedings page)",
     "url": "https://www.roboticsproceedings.org/rss21/p021.html",
     "type": "paper",
     "publisher": "RSS 2025",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "ICLR 2025, 7th Robot Learning Workshop, oral listing for ManiSkill3",
     "url": "https://iclr.cc/virtual/2025/32495",
     "type": "paper",
     "publisher": "ICLR 2025 workshop",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: mani-skill/ManiSkill (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/mani-skill/ManiSkill",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "ManiSkill GitHub releases and tags (v0.2.0 to v3.0.1)",
     "url": "https://api.github.com/repos/mani-skill/ManiSkill/releases",
     "type": "repo",
     "publisher": "mani-skill (UC San Diego Hao Su lab)",
     "date": "2026-04-21",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "PyPI: mani-skill release history (3.0.0.dev0 on 2024-03-09 to 3.0.1)",
     "url": "https://pypi.org/project/mani-skill/",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-04-21",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "ManiSkill README (licence section, RSS 2025, ManiSkill2 pointer)",
     "url": "https://github.com/mani-skill/ManiSkill/blob/main/README.md",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "ManiSkill LICENSE (Apache-2.0)",
     "url": "https://github.com/mani-skill/ManiSkill/blob/main/LICENSE",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "ManiSkill pull request #1269 '[Docs] Licensing' (adds LICENSE-3RD-PARTY, rewrites an LGPL-derived file)",
     "url": "https://github.com/mani-skill/ManiSkill/pull/1269",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2025-09-14",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "ManiSkill issue #1208 'Licensing mismatch'",
     "url": "https://github.com/mani-skill/ManiSkill/issues/1208",
     "type": "repo",
     "publisher": "mani-skill (community report)",
     "date": "2025-07-30",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "ManiSkill documentation: Tasks (index and 9 category pages; generated task tables)",
     "url": "https://maniskill.readthedocs.io/en/latest/tasks/index.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "ManiSkill documentation: Digital Twins (BridgeData v2 evaluation twins)",
     "url": "https://maniskill.readthedocs.io/en/latest/tasks/digital_twins/index.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "ManiSkill documentation: Reinforcement Learning Baselines ('Standard Benchmark')",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/reinforcement_learning/baselines.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "ManiSkill documentation: Reinforcement Learning Setup (evaluation protocol and metrics)",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/reinforcement_learning/setup.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "ManiSkill documentation: Learning from Demonstrations Setup (evaluation, success_once, backend pitfalls)",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/learning_from_demos/setup.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "ManiSkill documentation: Learning from Demonstrations Baselines (marked WIP)",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/learning_from_demos/baselines.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "ManiSkill commit bbf04bd4 'Update links to new github org' (changes the docs' Hugging Face demo link)",
     "url": "https://github.com/mani-skill/ManiSkill/commit/bbf04bd4",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-06-07",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Hugging Face dataset haosulab/ManiSkill_Demonstrations: card and file listing",
     "url": "https://huggingface.co/datasets/haosulab/ManiSkill_Demonstrations",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2025-07-06",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Hugging Face Hub API record for haosulab/ManiSkill_Demonstrations (downloads, likes)",
     "url": "https://huggingface.co/api/datasets/haosulab/ManiSkill_Demonstrations?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Hugging Face dataset mani-skill/ManiSkill_Demonstrations (link in the docs; HTTP 401)",
     "url": "https://huggingface.co/datasets/mani-skill/ManiSkill_Demonstrations",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "ManiSkill documentation: Scene Datasets",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/datasets/scenes.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "ManiSkill asset registry source (mani_skill/utils/assets/data.py)",
     "url": "https://github.com/mani-skill/ManiSkill/blob/main/mani_skill/utils/assets/data.py",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "ManiSkill robot asset folders with their own LICENSE files (allegro, dclaw, g1_humanoid, koch, so100)",
     "url": "https://github.com/mani-skill/ManiSkill/tree/main/mani_skill/assets/robots",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Hugging Face dataset cards for ManiSkill scene copies (haosulab/ReplicaCADRearrange, haosulab/AI2THOR, haosulab/RoboCasa)",
     "url": "https://huggingface.co/datasets/haosulab/RoboCasa",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Evaluating Real-World Robot Manipulation Policies in Simulation (SIMPLER), Table V and Figure 7",
     "url": "https://arxiv.org/abs/2405.05941",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "SimplerEnv README, maniskill3 branch",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/maniskill3/README.md",
     "type": "repo",
     "publisher": "SimplerEnv team",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "ManiSkill3 paper v2, HTML version, Figure 25 image (real vs sim success for octo_base, octo_small, rt-1x)",
     "url": "https://arxiv.org/html/2410.00425v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "ManiSkill issue #1380 'Significant discrepancy between CPU and GPU simulation backends'",
     "url": "https://github.com/mani-skill/ManiSkill/issues/1380",
     "type": "repo",
     "publisher": "mani-skill (user report, maintainer reply)",
     "date": "2026-01-25",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "ManiSkill pull request #1457 'Fix scalar-seeded main RNG expansion shrinking the episode seed batch'",
     "url": "https://github.com/mani-skill/ManiSkill/pull/1457",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-06-23",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "ManiSkill release v3.0.0b12 notes (RNG bug introduced in 3.0.0b11)",
     "url": "https://github.com/mani-skill/ManiSkill/releases/tag/v3.0.0b12",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2024-10-29",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "ManiSkill release v3.0.0b10 notes (real2sim twins added; evaluation metrics aligned)",
     "url": "https://github.com/mani-skill/ManiSkill/releases/tag/v3.0.0b10",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2024-10-01",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "ManiSkill release v3.0.1 notes",
     "url": "https://github.com/mani-skill/ManiSkill/releases/tag/v3.0.1",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-04-21",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "ManiSkill commit history, main branch",
     "url": "https://github.com/mani-skill/ManiSkill/commits/main",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026-08-02",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "GPUSimBench: Towards Scalable and Reliable GPU-Accelerated Simulators in Embodied AI (Table IV)",
     "url": "https://arxiv.org/abs/2607.13059",
     "type": "paper",
     "publisher": "arXiv (Shanghai AI Laboratory and others)",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Reliable and Scalable Robot Policy Evaluation with Imperfect Simulators (SureSim)",
     "url": "https://arxiv.org/abs/2510.04354",
     "type": "paper",
     "publisher": "arXiv (Princeton and others)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "Squint: Fast Visual Reinforcement Learning for Sim-to-Real Robotics",
     "url": "https://arxiv.org/abs/2602.21203",
     "type": "paper",
     "publisher": "arXiv (UC San Diego)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "RDT-1B README, 'Simulation Benchmark' section (ManiSkill five-task results)",
     "url": "https://github.com/thu-ml/RoboticsDiffusionTransformer/blob/main/README.md",
     "type": "repo",
     "publisher": "Tsinghua University (thu-ml)",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "E0: Enhancing Generalization and Fine-Grained Control in VLA Models via Tweedie Discrete Diffusion (Table 10)",
     "url": "https://arxiv.org/abs/2511.21542",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "What Can RL Bring to VLA Generalization? An Empirical Study (RL4VLA)",
     "url": "https://arxiv.org/abs/2505.19789",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "RLinf-VLA: A Unified and Efficient Framework for Reinforcement Learning of Vision-Language-Action Models",
     "url": "https://arxiv.org/abs/2510.06710",
     "type": "paper",
     "publisher": "arXiv (Tsinghua, Infinigence AI and others)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "piRL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models (ManiSkill benchmark, Table 4)",
     "url": "https://arxiv.org/abs/2510.25889",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "ManiSkill-HAB: A Benchmark for Low-Level Manipulation in Home Rearrangement Tasks",
     "url": "https://arxiv.org/abs/2412.13211",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "ManiSkill documentation: Community Projects",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/additional_resources/community_projects.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "ManiSkill documentation: Roadmap",
     "url": "https://maniskill.readthedocs.io/en/latest/roadmap/index.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "Semantic Scholar API record for arXiv:2410.00425",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2410.00425?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "ManiSkill: Generalizable Manipulation Skill Benchmark with Large-Scale Demonstrations (ManiSkill v1)",
     "url": "https://arxiv.org/abs/2107.14483",
     "type": "paper",
     "publisher": "arXiv (NeurIPS 2021 Datasets and Benchmarks)",
     "date": "2021-07",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "ManiSkill2: A Unified Benchmark for Generalizable Manipulation Skills",
     "url": "https://arxiv.org/abs/2302.04659",
     "type": "paper",
     "publisher": "arXiv (ICLR 2023)",
     "date": "2023-02",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "haosulab/ManiSkill-Legacy README (ManiSkill v1 code)",
     "url": "https://github.com/haosulab/ManiSkill-Legacy",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2022-08-24",
     "accessed": "2026-10-10"
    },
    "s51": {
     "title": "PyPI: mani-skill2 release history (0.5.3, 0.6.0.dev0 to dev3)",
     "url": "https://pypi.org/project/mani-skill2/",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2024-01-12",
     "accessed": "2026-10-10"
    },
    "s52": {
     "title": "TrossenRobotics/ManiSkill-WidowX_AI (asset files for ManiSkill)",
     "url": "https://github.com/TrossenRobotics/ManiSkill-WidowX_AI",
     "type": "repo",
     "publisher": "Trossen Robotics",
     "date": "2025-06-03",
     "accessed": "2026-10-10"
    },
    "s53": {
     "title": "ManiSkill release v3.0.0b22 notes (WidowX AI support, sim2real tooling with LeRobot hardware)",
     "url": "https://github.com/mani-skill/ManiSkill/releases/tag/v3.0.0b22",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2025-12-05",
     "accessed": "2026-10-10"
    },
    "s54": {
     "title": "ManiSkill documentation: Robots (35 robot pages)",
     "url": "https://maniskill.readthedocs.io/en/latest/robots/index.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s55": {
     "title": "ManiSkill demonstration metadata: PickCube-v1 motion-planning trajectory.json (1,000 episodes)",
     "url": "https://huggingface.co/datasets/haosulab/ManiSkill_Demonstrations/blob/main/demos/PickCube-v1/motionplanning/trajectory.json",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s57": {
     "title": "Semantic Scholar API records for arXiv:2107.14483 (ManiSkill) and arXiv:2302.04659 (ManiSkill2)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2302.04659?fields=citationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s58": {
     "title": "ManiSkill source: humanoid_pick_place.py (registers UnitreeG1PlaceAppleInBowl-v1; HumanoidPlaceAppleInBowl is a class name)",
     "url": "https://github.com/mani-skill/ManiSkill/blob/main/mani_skill/envs/tasks/humanoid/humanoid_pick_place.py",
     "type": "repo",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s56": {
     "title": "ManiSkill documentation: Vision Language Action Models",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/vision_language_action_models/index.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created full entry from primary sources, starting from the checked basic entry and the core-sim-a inventory notes. Corrected: robots (35 configurations in docs), scene (kitchen scenes have no scored tasks), leaderboard (paper-only), commercial use (non-commercial per README asset licence), real-to-sim study details (3 policies, real values taken from SIMPLER), latest commit 2026-08-02. Added lineage to ManiSkill 1 and 2, validity items, issues and readings."
    }
   ]
  },
  {
   "id": "meta-motivo-study",
   "name": "Meta Motivo study",
   "full_name": "Meta Motivo human-likeness preference study",
   "aliases": [
    "Meta Motivo human evaluation",
    "FB-CPR vs TD3 human study"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "A one-off human evaluation inside a model paper (ICLR 2025), comparing two policies on a simulated SMPL human avatar, not a robot. Not designed to be re-run; rater data not released. Closest to the 'virtual agents' borderline case.",
   "summary": {
    "text": "Fifty raters compared videos of two simulated-humanoid policies on 95 tasks for task success and human-likeness.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "FAIR at Meta; one author at Mila/McGill (work done at Meta), one at FAIR and UCL.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Authors' office locations not stated."
    },
    "first_release": {
     "value": "2024-12-12 (Meta research publication page).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "No change to the study after arXiv v1 (2025-04-15). Code repo last push 2025-06-10 (bug fix).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "arXiv v1; code release tags v0.1.0-v0.1.2 (latest 2025-01-27).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "50 human raters; all 45 reward tasks and 50 goal tasks; two agents compared (TD3 best of 3 seeds vs zero-shot FB-CPR); video positions randomised, agent identity hidden.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "preference",
      "human-rating"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Not an Elo or Bradley-Terry ranking."
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (None; results only in paper figures.)"
    },
    "license_code": {
     "value": "CC BY-NC 4.0 (LICENSE file of facebookresearch/metamotivo).",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked at arXiv abs links, paper text, repo top-level listing."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "Not checked (None found; no physical robot in the paper.)",
     "note": "Body is a simulated SMPL humanoid character in MuJoCo; no physical counterpart and no real-robot experiment in the paper."
    },
    "capability": {
     "value": [
      "human-likeness",
      "locomotion"
     ],
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "kind": {
     "value": "study",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models (full text)",
     "url": "https://arxiv.org/html/2504.11054v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-04"
    },
    "s2": {
     "title": "Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models | Research - AI at Meta",
     "url": "https://ai.meta.com/research/publications/zero-shot-whole-body-humanoid-control-via-behavioral-foundation-models/",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models",
     "url": "https://arxiv.org/abs/2504.11054",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-04"
    },
    "s4": {
     "title": "facebookresearch/metamotivo on GitHub (repository)",
     "url": "https://github.com/facebookresearch/metamotivo",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "facebookresearch/metamotivo on GitHub (blob)",
     "url": "https://github.com/facebookresearch/metamotivo/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2504.11054",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "meta-world",
   "name": "Meta-World",
   "full_name": "Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning",
   "aliases": [
    "MetaWorld",
    "Meta-World+",
    "Meta-World v3",
    "MT1",
    "MT10",
    "MT25",
    "MT50",
    "ML1",
    "ML10",
    "ML25",
    "ML45",
    "MT10-rand"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "Meta-World is a simulated benchmark of 50 tabletop tasks for one Sawyer arm, built to test multi-task and meta reinforcement learning and scored by success rate. It is maintained by the Farama Foundation; recent VLA papers also report it with their own averaging rule.",
    "short": "Meta-World is a set of 50 simulated tabletop tasks for one Sawyer robot arm. It is used to test multi-task and meta reinforcement learning (learning to adapt quickly to new tasks), and recent vision-language-action (VLA) papers also report results on it.",
    "sources": [
     "s2",
     "s4",
     "s9"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "A suite of 50 tasks with defined multi-task and meta-learning evaluation protocols."
    },
    "publishers": {
     "value": [
      "Stanford University",
      "UC Berkeley",
      "Columbia University",
      "University of Southern California",
      "Robotics at Google",
      "Farama Foundation (maintainer since 2023)"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s6",
      "s4"
     ],
     "checked": "2026-10-11",
     "items": [
      {
       "value": "Original authors (2019)",
       "display": "Tianhe Yu, Deirdre Quillen, Zhanpeng He, Ryan Julian, Avnish Narayan, Hayden Shively, Adithya Bellathur, Karol Hausman, Chelsea Finn, Sergey Levine: Stanford, UC Berkeley, Columbia, USC, Robotics at Google.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Farama Foundation",
       "display": "Maintains the code since 2023; release v2.0.0 (2023-06-16) is 'the code before the Farama Foundation started maintaining Meta-World'.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "Meta-World+ authors (2025)",
       "display": "Toronto Metropolitan University, University of Surrey, Hamburg University of Technology, Columbia, Google DeepMind, USC, Farama Foundation, Université de Montréal / Mila.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-11",
     "note": "Built by university labs with Robotics at Google; now maintained by the Farama Foundation, a non-profit open-source foundation."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Original institutions in the United States."
    },
    "first_release": {
     "value": "2019-10",
     "display": "arXiv v1 on 2019-10-24; CoRL 2019. The paper was updated as arXiv v2 on 2021-06-14.",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-11",
     "note": "GitHub repository created 2019-09-09.",
     "short": "October 2019, at CoRL 2019"
    },
    "published_at": {
     "value": "CoRL 2019 (PMLR 100); Meta-World+ at NeurIPS 2025 Datasets and Benchmarks",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s9"
     ],
     "checked": "2026-10-11",
     "short": "CoRL 2019, and NeurIPS 2025 Datasets and Benchmarks for Meta-World+"
    },
    "latest_update": {
     "value": "2026-10",
     "display": "Last commit 2026-10-09 (drop Python 3.10). Last release v3.1.1 on 2026-06-28.",
     "level": "verified",
     "sources": [
      "s7",
      "s6",
      "s5"
     ],
     "checked": "2026-10-11",
     "short": "October 2026. The last release was v3.1.1 in June 2026."
    },
    "version": {
     "value": "3.1.1",
     "level": "verified",
     "sources": [
      "s6",
      "s12"
     ],
     "checked": "2026-10-11",
     "note": "Environment names now end in '-v3'.",
     "items": [
      {
       "value": "v2.0.0 (2023-06-16)",
       "display": "Code as it was before Farama took over maintenance.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "3.0.0 (2025-06-14)",
       "display": "MuJoCo Python bindings replace mujoco-py; Gymnasium 1.0 API; the original 'V1' reward functions exposed again as an option; V1 environments removed.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "v3.1.1 (2026-06-28)",
       "display": "Pins mujoco==3.3.0 because newer MuJoCo changes how contacts are represented; fixes button-press dense rewards; adds a camera. Release notes say v3.1.0 was skipped, but PyPI lists 3.1.0 (uploaded 2026-06-26).",
       "level": "verified",
       "sources": [
        "s6",
        "s12"
       ]
      },
      {
       "value": "Meta-World+ (paper, 2025)",
       "display": "Describes the v3 changes, adds MT25 and ML25 task sets and custom task sets.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      }
     ],
     "short": "3.1.1, released June 2026"
    },
    "capability": {
     "value": [
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "50 tabletop tasks such as reach, push, pick-place, door and drawer opening and peg insertion. The benchmark's purpose, multi-task and meta reinforcement learning, has no taxonomy value."
    },
    "generalisation": {
     "value": [
      "object-pose",
      "new-task"
     ],
     "display": "Depends on the protocol. MT1, MT10 and MT50 test on the goal positions seen in training. ML1 tests new goal positions; ML10 and ML45 test 5 held-out tasks.",
     "level": "verified",
     "sources": [
      "s10",
      "s2"
     ],
     "checked": "2026-10-11",
     "note": "The docs say multi-task agents are evaluated on 'the same ones seen during training'.",
     "short": "New goal positions in ML1, or 5 held-out tasks in ML10 and ML45"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "MuJoCo 3.3.0 (pinned in v3.1.1)",
     "level": "verified",
     "sources": [
      "s2",
      "s6",
      "s12"
     ],
     "checked": "2026-10-11",
     "note": "Originally mujoco-py; official MuJoCo bindings since 3.0.0.",
     "short": "MuJoCo"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Sawyer (simulated)",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Actions: 3D end-effector displacement plus a gripper value. Observations: 39-dimensional state including object and goal positions; goal zeroed for meta-learning."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s14"
     ],
     "checked": "2026-10-11"
    },
    "tasks": {
     "value": 50,
     "display": "50 tasks; task sets MT1, MT10, MT50, ML1, ML10, ML45, plus MT25 and ML25 since 2025",
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-11",
     "note": "The repository has 50 '_v3' task environment files (counted by us).",
     "short": "50 tasks"
    },
    "scenes": {
     "value": 1,
     "display": "One shared tabletop scene; tasks differ in objects",
     "level": "inferred",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-11",
     "short": "1 scene"
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-11",
     "note": "No object count in the paper, docs or README. The assets folder holds 42 object model files and 44 Sawyer scene files (counted by us); file counts are not object counts."
    },
    "demonstrations": {
     "value": "none (scripted experts)",
     "display": "No official dataset. Tasks have scripted expert policies; imitation-learning papers generate their own demonstrations (often 50 per task).",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s28",
      "s15"
     ],
     "checked": "2026-10-11",
     "items": [
      {
       "value": "lerobot/metaworld_mt50",
       "display": "Hugging Face dataset (Apache-2.0 tag) used by SmolVLA: 2,500 episodes, 50 per task; 5,854 recent downloads.",
       "level": "verified",
       "sources": [
        "s15",
        "s18"
       ]
      }
     ],
     "short": "There is no fixed dataset. Each task has a scripted expert policy."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s10",
      "s2"
     ],
     "checked": "2026-10-11",
     "note": "Rewards are for training; evaluation uses a per-task success flag, usually an object within a distance threshold (for example 5 cm) of its goal."
    },
    "metric_detail": {
     "value": "Mean success rate (%)",
     "display": "An episode counts as a success if the success flag is raised at any point. Multi-task RL: one episode per training goal (50 goals) per task, averaged. Meta-RL: 3 episodes per test goal per test task after adaptation. Imitation-learning (VLA) papers: 10 episodes for each of the 50 tasks, then the plain mean of four difficulty-tier averages.",
     "level": "verified",
     "sources": [
      "s10",
      "s4",
      "s18",
      "s24"
     ],
     "checked": "2026-10-11",
     "note": "The tier grouping (easy 28, medium 11, hard 6, very hard 5) comes from Seo et al. (2023), as cited by SmolVLA and TinyVLA.",
     "short": "Success rate. The protocol varies between papers."
    },
    "trials": {
     "value": "50 episodes per task (RL); 10 per task (VLA papers)",
     "level": "verified",
     "sources": [
      "s10",
      "s14",
      "s18",
      "s24",
      "s4"
     ],
     "checked": "2026-10-11",
     "items": [
      {
       "value": "Official evaluation utility",
       "display": "Multi-task: one episode per each of 50 goal positions per task, 500-step horizon. Meta: 40 test goals per test task, 3 episodes each after adaptation.",
       "level": "verified",
       "sources": [
        "s10"
       ]
      },
      {
       "value": "Meta-World+ experiments",
       "display": "10 seeds, interquartile mean with 95% confidence intervals.",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "LeRobot",
       "display": "Recommends 10 episodes per task (500 for MT50); default evaluation on the 'medium' group.",
       "level": "verified",
       "sources": [
        "s14"
       ]
      },
      {
       "value": "FabriVLA",
       "display": "10 episodes per task, 400-step horizon.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      }
     ],
     "short": "50 per task in RL papers, 10 per task in VLA papers"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s26",
      "s18",
      "s24"
     ],
     "checked": "2026-10-11",
     "note": "Meta-World+ reports IQM with 95% CIs over 10 seeds and MOORE reports standard deviations. The original paper's Table 1 gives means over 10 seeds without spread. SmolVLA, Evo-1 and FabriVLA give single numbers."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s9",
      "s33"
     ],
     "checked": "2026-10-11",
     "note": "No submission process or organiser evaluation."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s9",
      "s33",
      "s4"
     ],
     "checked": "2026-10-11",
     "note": "No leaderboard on the docs site, README or Meta-World+ paper. Meta-World+ advises users to run their own baselines rather than copy numbers."
    },
    "top_score": {
     "value": 90,
     "display": "90.0% (FabriVLA, July 2026) on the imitation-learning protocol used by recent VLA papers: 50 tasks, mean of four difficulty-tier averages. RL protocols are reported separately (items). Rows from different papers use different training data and trials.",
     "level": "inferred",
     "sources": [
      "s24",
      "s23"
     ],
     "checked": "2026-10-11",
     "note": "Each score is verified at its source. 'Highest' is our judgement from the papers we opened and one web search on 2026-10-10. The chart series uses the tier-average protocol because recent model reports use it; MT10, MT50, ML10 and ML45 results from RL papers are a different measurement.",
     "items": [
      {
       "value": 10.5,
       "display": "Diffusion Policy, as run by the TinyVLA authors, 2024-09: 10.5 (tier average)",
       "level": "verified",
       "sources": [
        "s17"
       ],
       "data": {
        "model": "Diffusion Policy",
        "date": "2024-09",
        "avg": 10.5,
        "rl": false
       }
      },
      {
       "value": 31.6,
       "display": "TinyVLA-H, 2024-09: 77.6 / 21.5 / 11.4 / 15.8 → 31.6",
       "level": "verified",
       "sources": [
        "s17"
       ],
       "note": "3 seeds, 50 demonstrations per task.",
       "data": {
        "model": "TinyVLA-H",
        "date": "2024-09",
        "avg": 31.6,
        "rl": false,
        "tiers": [
         77.6,
         21.5,
         11.4,
         15.8
        ]
       }
      },
      {
       "value": 47.9,
       "display": "π0 (3.5B, robot-pretrained), run by the SmolVLA team, 2025-06: 71.8 / 48.2 / 41.7 / 30.0 → 47.9",
       "level": "verified",
       "sources": [
        "s18"
       ],
       "data": {
        "model": "π0 (run by SmolVLA team)",
        "date": "2025-06",
        "avg": 47.9,
        "rl": false,
        "tiers": [
         71.8,
         48.2,
         41.7,
         30
        ]
       }
      },
      {
       "value": 68.24,
       "display": "SmolVLA (2.25B), 2025-06: 87.14 / 51.82 / 70 / 64 → 68.24",
       "level": "verified",
       "sources": [
        "s18"
       ],
       "note": "The released 0.45B model scores 57.3. 2,500 demonstrations; 10 trials per task.",
       "data": {
        "model": "SmolVLA (2.25B)",
        "date": "2025-06",
        "avg": 68.24,
        "rl": false,
        "tiers": [
         87.14,
         51.82,
         70,
         64
        ]
       }
      },
      {
       "value": 85.8,
       "display": "πRL on π0 (Flow-Noise), 2025-10: 91.1 / 81.8 / 78.3 / 92.0 → 85.8 after RL in the simulator",
       "level": "verified",
       "sources": [
        "s19"
       ],
       "note": "Supervised starting point 50.8.",
       "data": {
        "model": "πRL on π0 (Flow-Noise)",
        "date": "2025-10",
        "avg": 85.8,
        "rl": true,
        "tiers": [
         91.1,
         81.8,
         78.3,
         92
        ]
       }
      },
      {
       "value": 80.6,
       "display": "Evo-1 (0.77B), 2025-11: 89.2 / 76.8 / 77.2 / 79.2 → 80.6",
       "level": "verified",
       "sources": [
        "s20"
       ],
       "note": "10 trials per task.",
       "data": {
        "model": "Evo-1",
        "date": "2025-11",
        "avg": 80.6,
        "rl": false,
        "tiers": [
         89.2,
         76.8,
         77.2,
         79.2
        ]
       }
      },
      {
       "value": 67.85,
       "display": "AnoleVLA (0.47B), 2026-03: 89.29 / 45.45 / 66.67 / 70.00 → 67.85",
       "level": "verified",
       "sources": [
        "s21"
       ],
       "data": {
        "model": "AnoleVLA",
        "date": "2026-03",
        "avg": 67.85,
        "rl": false,
        "tiers": [
         89.29,
         45.45,
         66.67,
         70
        ]
       }
      },
      {
       "value": 84.4,
       "display": "Evo-Depth (0.9B), 2026-05: 83.1 / 84.7 / 87.3 / 82.4 → 84.4",
       "level": "verified",
       "sources": [
        "s22"
       ],
       "data": {
        "model": "Evo-Depth",
        "date": "2026-05",
        "avg": 84.4,
        "rl": false,
        "tiers": [
         83.1,
         84.7,
         87.3,
         82.4
        ]
       }
      },
      {
       "value": 87.53,
       "display": "LA4VLA (1B), 2026-06: 88.9 / 94.5 / 66.7 / 100.0 → 87.53",
       "level": "verified",
       "sources": [
        "s23"
       ],
       "data": {
        "model": "LA4VLA",
        "date": "2026-06",
        "avg": 87.53,
        "rl": false,
        "tiers": [
         88.9,
         94.5,
         66.7,
         100
        ]
       }
      },
      {
       "value": 90,
       "display": "FabriVLA (0.88B), 2026-07: 95.0 / 88.2 / 86.7 / 90.0 → 90.0",
       "level": "verified",
       "sources": [
        "s24"
       ],
       "note": "10 episodes per task; 400-step horizon. The paper also reports 92.0% over all episodes.",
       "data": {
        "model": "FabriVLA",
        "date": "2026-07",
        "avg": 90,
        "rl": false,
        "tiers": [
         95,
         88.2,
         86.7,
         90
        ]
       }
      },
      {
       "value": "RL protocols (not in chart)",
       "display": "Original paper (10 seeds, average of maximum success): MT10 68.3% and MT50 38.5% (multi-task SAC); ML10 meta-test 35.8% (RL2); ML45 meta-test 39.9% (MAML). MOORE (ICLR 2024): MT10-rand 88.7 ± 5.6, MT50-rand 72.9 ± 3.3. Meta-World+ re-runs with V2 rewards: MT10 85.52 ± 1.72 (PCGrad), MT50 71.99 ± 2.93 (MOORE), ML45 36.76 ± 13.90 (RL2).",
       "level": "verified",
       "sources": [
        "s2",
        "s26",
        "s4"
       ]
      }
     ],
     "short": "90.0% average over four difficulty tiers (July 2026)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s8",
      "s12"
     ],
     "checked": "2026-10-11",
     "note": "LICENSE: MIT, Copyright (c) 2019 Meta-World Team. PyPI metadata: MIT License."
    },
    "license_data": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s9",
      "s15"
     ],
     "checked": "2026-10-11",
     "note": "No dataset distributed by the maintainers. The third-party LeRobot copy lerobot/metaworld_mt50 carries an Apache-2.0 tag."
    },
    "license_assets": {
     "value": "MIT",
     "level": "inferred",
     "sources": [
      "s13",
      "s8"
     ],
     "checked": "2026-10-11",
     "note": "The repository's only licence file is the root MIT LICENSE; the robot and object model files sit in the same repository. Their original sources are not stated. Not legal advice."
    },
    "access": {
     "value": "open",
     "display": "pip install metaworld; no registration.",
     "level": "verified",
     "sources": [
      "s9",
      "s12"
     ],
     "checked": "2026-10-11",
     "short": "Open. It installs with pip."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s8",
      "s13"
     ],
     "checked": "2026-10-11",
     "note": "MIT licence for code and bundled assets; no dataset. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Some models scored on Meta-World also ran on real robots (for example SmolVLA on SO-100 arms and FabriVLA on a Unitree D1-T arm), on different tasks. No published study compares the same policies' Meta-World and real-robot scores.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s18",
      "s24",
      "s31",
      "s34"
     ],
     "checked": "2026-10-11",
     "note": "Neither the original paper nor Meta-World+ has real-robot experiments. PolaRiS cites Meta-World among simulation benchmarks that fail to capture real-world visual complexity, without measuring it. The basic entry graded this 'none-found'; we use 'demonstrated' because the taxonomy defines it as 'some policies also ran on real robots'.",
     "short": "No study has compared its scores with real-robot scores for the same policies."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "Continual World",
      "MTBench"
     ],
     "level": "verified",
     "sources": [
      "s29",
      "s30"
     ],
     "checked": "2026-10-11",
     "note": "Not a complete list. Meta-World+ is a new version of Meta-World itself, recorded under facts.version.",
     "items": [
      {
       "value": "Continual World",
       "display": "NeurIPS 2021. Continual-RL benchmark built on Meta-World tasks. 154 citations (Semantic Scholar).",
       "level": "verified",
       "sources": [
        "s29",
        "s16"
       ]
      },
      {
       "value": "MTBench",
       "display": "RLC 2025. Re-implements Meta-World's 50 tasks (plus 20 locomotion tasks) in the GPU simulator IsaacGym.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      }
     ],
     "short": "2 benchmarks built on Meta-World"
    },
    "citations": {
     "value": 1843,
     "display": "1,843 (Semantic Scholar; 326 influential). Meta-World+: 58.",
     "level": "verified",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "short": "1,843"
    },
    "github_stars": {
     "value": 1890,
     "display": "1,890 stars, 357 forks, 19 open issues (Farama-Foundation/Metaworld)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-11",
     "short": "1,890"
    },
    "used_by": {
     "value": "Used by multi-task and meta-RL papers since 2019 and by lightweight VLA reports since 2024; we found no count of papers.",
     "level": "verified",
     "sources": [
      "s4",
      "s18",
      "s27",
      "s28",
      "s20",
      "s32"
     ],
     "checked": "2026-10-11",
     "note": "The 2026 audit cites Meta-World but did not audit it; it chose the five benchmarks most reported in recent VLA work.",
     "items": [
      {
       "value": "Gato",
       "display": "DeepMind, 2022: trained on 45 Meta-World tasks (94.6K episodes) and reports over 50% success on 44 of them.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "SmolVLA",
       "display": "Hugging Face, 2025-06",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "Evo-1, Evo-Depth, LA4VLA",
       "display": "Shanghai Jiao Tong University and EvoMind Tech, 2025-2026",
       "level": "verified",
       "sources": [
        "s20",
        "s22",
        "s23"
       ]
      },
      {
       "value": "πRL",
       "display": "RL fine-tuning of π0 and π0.5, 2025-10",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "Rho",
       "display": "Microsoft Research, 2026-09: latent-space adaptation on six Meta-World tasks (average 57.5% to 73.3%, 100 rollouts per task)",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "MOORE, PaCo, Soft Modularization, PCGrad",
       "display": "Multi-task RL methods compared in Meta-World+",
       "level": "verified",
       "sources": [
        "s4",
        "s26"
       ]
      }
     ],
     "short": "RL papers and small VLA models"
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Hugging Face",
      "Microsoft Research",
      "EvoMind Tech"
     ],
     "level": "verified",
     "sources": [
      "s27",
      "s4",
      "s14",
      "s18",
      "s28",
      "s20"
     ],
     "checked": "2026-10-11",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "Gato trained on Meta-World; Google DeepMind researchers co-authored Meta-World+.",
       "level": "verified",
       "sources": [
        "s27",
        "s4"
       ]
      },
      {
       "value": "Hugging Face",
       "display": "LeRobot ships a Meta-World environment (added 2025-10) and the metaworld_mt50 dataset; SmolVLA reports Meta-World.",
       "level": "verified",
       "sources": [
        "s14",
        "s15",
        "s18"
       ]
      },
      {
       "value": "Microsoft Research",
       "display": "Rho uses Meta-World for adaptation experiments.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "EvoMind Tech",
       "display": "Affiliation on Evo-1, which reports 80.6% on Meta-World.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      }
     ]
    },
    "status": {
     "value": "active",
     "display": "Commits in October 2026; release in June 2026.",
     "level": "inferred",
     "sources": [
      "s7",
      "s6",
      "s5"
     ],
     "checked": "2026-10-11",
     "note": "Within the six-month rule.",
     "short": "Active. There were commits in October 2026."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Two reward versions made published results incomparable",
     "text": "Meta-World+ reports that the original 'V1' dense rewards were overwritten by 'V2' rewards without documentation, so papers used different rewards. In its re-runs every multi-task method scored higher with V2: PaCo reached 26.2% on MT10 with V1 and 73.6% with V2. Published numbers disagree too: PaCo's MT10 is 71.6 in its own paper and 85.4 as reported by MOORE; PCGrad's published 90% on MT10 could not be reproduced (70.5%), likely because it used a Meta-World version no longer available.",
     "level": "verified",
     "sources": [
      "s4",
      "s26"
     ],
     "status": "open",
     "mitigation": {
      "text": "Since 3.0.0 (2025-06) both reward versions can be selected, and Meta-World+ asks authors to state the version and run their own baselines.",
      "sources": [
       "s6",
       "s4"
      ]
     },
     "short": "The original V1 rewards (the feedback signal used in reinforcement learning) were replaced by V2 rewards without documentation. The two versions give different scores, so older results cannot be compared."
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "Protocol details differ between the paper, the docs and later papers",
     "text": "The original paper and the benchmark docs say MT10 and MT50 goal positions are fixed (the docs carry a 'TODO: check this'), while the evaluation docs cycle through 50 goal positions per task; MOORE and PaCo report 'MT10-rand' with random goals. ML1 is described with 50 held-out positions (paper), 10 (benchmark docs) and 40 (evaluation docs). The original paper reports the average of the maximum success reached during training. Horizons differ (500 steps in the docs, 400 in FabriVLA).",
     "level": "verified",
     "sources": [
      "s2",
      "s11",
      "s10",
      "s26",
      "s24"
     ],
     "status": "open",
     "short": "Sources disagree on whether goal positions are fixed or random. They also give different numbers of test goals."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "VLA papers average four difficulty tiers instead of 50 tasks",
     "text": "Recent VLA papers report the plain mean of four difficulty-tier averages (28 easy, 11 medium, 6 hard, 5 very hard tasks), so each very hard task weighs about 5.6 times as much as an easy one. For SmolVLA (0.45B) the tier mean is 57.3%, while a task-weighted mean of the same tiers is about 66.8% (our arithmetic). The same model also appears with different numbers: π0 at 47.9, 50.5, 50.8 and 47.91; SmolVLA at 57.3 or 68.2 depending on model size.",
     "level": "inferred",
     "sources": [
      "s18",
      "s24",
      "s19",
      "s25",
      "s21"
     ],
     "status": "open",
     "note": "The tier counts and averaging rule are verified (FabriVLA, LeRobot docs); the weighting comparison is our arithmetic.",
     "short": "VLA papers take the plain mean of four difficulty-tier averages. This gives the 5 very hard tasks as much weight as the 28 easy ones."
    },
    {
     "id": "i4",
     "type": "protocol-variance",
     "title": "Results depend on the MuJoCo version",
     "text": "v3.1.1 pins mujoco==3.3.0 because newer MuJoCo versions change how contacts between objects are represented. The same release fixed the button-press dense rewards. Results under other MuJoCo versions, or before the fix, may differ.",
     "level": "verified",
     "sources": [
      "s6",
      "s12"
     ],
     "status": "open",
     "short": "Newer versions of MuJoCo (the physics simulator) change how contacts between objects are represented. For this reason, v3.1.1 requires MuJoCo 3.3.0."
    },
    {
     "id": "i5",
     "type": "saturated",
     "title": "Scores on some protocols are approaching 100%",
     "text": "On the VLA tier-average protocol, reported scores rose from 31.6% (TinyVLA, 2024-09) to 90.0% (FabriVLA, 2026-07), with individual tiers at up to 100%. MT10 reaches 88.7% (MOORE, 2024). Meta-learning remains far from the ceiling: Meta-World+ re-runs with V2 rewards give 25.7% to 36.76% on ML10 and ML45 (MAML and RL2).",
     "level": "verified",
     "sources": [
      "s17",
      "s24",
      "s23",
      "s26",
      "s4"
     ],
     "status": "open",
     "short": "In VLA papers, the average over the four tiers reaches 90%. Meta-learning scores stay near 30% to 40%."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A 'Meta-World score' can come from multi-task reinforcement learning (RL) with V1 or V2 rewards, from meta-RL on held-out tasks, or from an imitation-learning tier average. These are different measurements. Compare numbers only when they use the same protocol, version and averaging rule.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "issues.i3",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "short": "Compare numbers only when they use the same protocol and version."
    },
    {
     "id": "r2",
     "text": "A high score shows that a policy can handle 50 short tabletop tasks in one fixed simulated scene, usually on goal positions seen in training. It says little about performance on real robots or about how well the policy copes with visual changes.",
     "basis": [
      "facts.generalisation",
      "facts.sim_to_real",
      "facts.scenes"
     ],
     "confidence": "medium",
     "short": "A high score shows that a policy can do these tasks in one fixed simulated scene."
    },
    {
     "id": "r3",
     "text": "Meta-World is actively maintained and has numbered versions. This makes it a reasonable choice for RL research that needs reproducible results, as long as papers state the package version, the reward version and the MuJoCo version.",
     "basis": [
      "facts.status",
      "facts.version",
      "issues.i4"
     ],
     "confidence": "medium",
     "short": "Meta-World is actively maintained. Papers should state the versions they used so that others can reproduce their results."
    },
    {
     "id": "r4",
     "text": "For VLA results, read the score for each difficulty tier. The headline average gives the 5 very hard tasks as much weight as the 28 easy ones. Small gains in the average can come from a few tasks.",
     "basis": [
      "issues.i3",
      "facts.top_score"
     ],
     "confidence": "medium",
     "short": "Read the score for each tier as well as the average."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy (the robot's control model) will do on a real robot.",
     "sub": "We found no study that compares Meta-World scores with real-robot scores for the same policies.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy copes with visual changes.",
     "sub": "All tasks use one fixed scene. Policies are usually tested on goal positions seen in training.",
     "basis": [
      "facts.scenes",
      "facts.generalisation"
     ]
    },
    {
     "id": "l3",
     "text": "Whether results from different papers can be compared.",
     "sub": "Papers use different reward versions and different rules for averaging scores.",
     "basis": [
      "issues.i1",
      "issues.i3"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Meta-World paper v2 and Meta-World+ (no real-robot experiments); README and docs; PolaRiS (cites Meta-World, no measurement); X2Real (related work only); SmolVLA and FabriVLA (real-robot tests on separate tasks). A dedicated web search could not run on 2026-10-11 because the shared search budget was used up; the earlier inventory search (2026-10-10) also found no paired study.",
     "date": "2026-10-11"
    },
    {
     "for": "leaderboard",
     "where": "README, docs site, Meta-World+ paper.",
     "date": "2026-10-11"
    },
    {
     "for": "top_score",
     "where": "One web search for 2026 tier-average results (2026-10-10), then the papers listed in top_score; FabriVLA's comparison table.",
     "date": "2026-10-11"
    },
    {
     "for": "license_assets",
     "where": "Repository tree (only the root LICENSE), asset XML headers.",
     "date": "2026-10-11"
    }
   ],
   "sources": {
    "s1": {
     "title": "Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning (arXiv abstract page; v1 2019-10-24, v2 2021-06-14)",
     "url": "https://arxiv.org/abs/1910.10897",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2019-10",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Meta-World paper, full text v2 (updated version of the CoRL 2019 paper)",
     "url": "https://arxiv.org/pdf/1910.10897",
     "type": "paper",
     "publisher": "arXiv (Stanford, UC Berkeley, Columbia, USC, Robotics at Google)",
     "date": "2021-06",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Meta-World+: An Improved, Standardized, RL Benchmark (arXiv abstract page; v1 2025-05-16, v2 2025-11-21)",
     "url": "https://arxiv.org/abs/2505.11289",
     "type": "paper",
     "publisher": "NeurIPS 2025 Datasets and Benchmarks",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Meta-World+ full text v2",
     "url": "https://arxiv.org/pdf/2505.11289",
     "type": "paper",
     "publisher": "NeurIPS 2025 Datasets and Benchmarks (Toronto Metropolitan University, Farama Foundation, University of Surrey, Google DeepMind and others)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "GitHub API: Farama-Foundation/Metaworld (stars, forks, open issues, pushed)",
     "url": "https://api.github.com/repos/Farama-Foundation/Metaworld",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-11",
     "accessed": "2026-10-11"
    },
    "s6": {
     "title": "Meta-World releases (v2.0.0, 3.0.0, v3.1.1) with release notes",
     "url": "https://github.com/Farama-Foundation/Metaworld/releases",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026-06-28",
     "accessed": "2026-10-11"
    },
    "s7": {
     "title": "Meta-World commit history",
     "url": "https://github.com/Farama-Foundation/Metaworld/commits/main",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026-10-09",
     "accessed": "2026-10-11"
    },
    "s8": {
     "title": "Meta-World LICENSE (MIT, Copyright (c) 2019 Meta-World Team)",
     "url": "https://github.com/Farama-Foundation/Metaworld/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2019",
     "accessed": "2026-10-11"
    },
    "s9": {
     "title": "Meta-World README",
     "url": "https://github.com/Farama-Foundation/Metaworld/blob/main/README.md",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026",
     "accessed": "2026-10-11"
    },
    "s10": {
     "title": "Meta-World docs: Evaluation (evaluation.md)",
     "url": "https://github.com/Farama-Foundation/Metaworld/blob/main/docs/evaluation/evaluation.md",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026",
     "accessed": "2026-10-11"
    },
    "s11": {
     "title": "Meta-World docs: Benchmark Descriptions (benchmark_descriptions.md)",
     "url": "https://github.com/Farama-Foundation/Metaworld/blob/main/docs/benchmark/benchmark_descriptions.md",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026",
     "accessed": "2026-10-11"
    },
    "s12": {
     "title": "PyPI: metaworld (versions, licence, mujoco==3.3.0 requirement)",
     "url": "https://pypi.org/pypi/metaworld/json",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026-06-28",
     "accessed": "2026-10-11"
    },
    "s13": {
     "title": "Meta-World repository tree (assets folders; licence files)",
     "url": "https://github.com/Farama-Foundation/Metaworld/tree/main/metaworld/assets",
     "type": "repo",
     "publisher": "Farama Foundation",
     "date": "2026",
     "accessed": "2026-10-11"
    },
    "s14": {
     "title": "LeRobot documentation: Meta-World (metaworld.mdx) and its commit history",
     "url": "https://github.com/huggingface/lerobot/blob/main/docs/source/metaworld.mdx",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-06",
     "accessed": "2026-10-11"
    },
    "s15": {
     "title": "Hugging Face Hub API record for lerobot/metaworld_mt50 (licence tag, downloads)",
     "url": "https://huggingface.co/api/datasets/lerobot/metaworld_mt50",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-11",
     "accessed": "2026-10-11"
    },
    "s16": {
     "title": "Semantic Scholar batch API records (ARXIV:1910.10897, 2505.11289, 2105.10919, 2507.23172)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "TinyVLA: Towards Fast, Data-Efficient Vision-Language-Action Models for Robotic Manipulation",
     "url": "https://arxiv.org/abs/2409.12514",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-09",
     "accessed": "2026-10-11"
    },
    "s18": {
     "title": "SmolVLA: A vision-language-action model for affordable and efficient robotics (Table 2)",
     "url": "https://arxiv.org/abs/2506.01844",
     "type": "paper",
     "publisher": "Hugging Face",
     "date": "2025-06",
     "accessed": "2026-10-11"
    },
    "s19": {
     "title": "πRL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models (v3, Table 6)",
     "url": "https://arxiv.org/abs/2510.25889",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Evo-1: Lightweight Vision-Language-Action Model with Preserved Semantic Alignment",
     "url": "https://arxiv.org/abs/2511.04555",
     "type": "paper",
     "publisher": "arXiv (Shanghai Jiao Tong University, EvoMind Tech and others)",
     "date": "2025-11",
     "accessed": "2026-10-11"
    },
    "s21": {
     "title": "AnoleVLA: Lightweight Vision-Language-Action Model with Deep State Space Models for Mobile Manipulation",
     "url": "https://arxiv.org/abs/2603.15046",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-03",
     "accessed": "2026-10-11"
    },
    "s22": {
     "title": "Evo-Depth: A Lightweight Depth-Enhanced Vision-Language-Action Model",
     "url": "https://arxiv.org/abs/2605.14950",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-11"
    },
    "s23": {
     "title": "LA4VLA: Learning to Act without Seeing via Language-Action Pretraining",
     "url": "https://arxiv.org/abs/2606.27295",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-11"
    },
    "s24": {
     "title": "FabriVLA: A Lightweight Vision-Language-Action Model with Conformal Action Chunk Uncertainty (v3)",
     "url": "https://arxiv.org/abs/2607.08575",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-07",
     "accessed": "2026-10-11"
    },
    "s25": {
     "title": "One Token Per Frame: Reconsidering Visual Bandwidth in World Models for VLA Policy (OneWM)",
     "url": "https://arxiv.org/abs/2605.07931",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-11"
    },
    "s26": {
     "title": "Multi-Task Reinforcement Learning with Mixture of Orthogonal Experts (MOORE)",
     "url": "https://arxiv.org/abs/2311.11385",
     "type": "paper",
     "publisher": "ICLR 2024",
     "date": "2023-11",
     "accessed": "2026-10-11"
    },
    "s27": {
     "title": "A Generalist Agent (Gato)",
     "url": "https://arxiv.org/abs/2205.06175",
     "type": "paper",
     "publisher": "DeepMind",
     "date": "2022-05",
     "accessed": "2026-10-11"
    },
    "s28": {
     "title": "Rho: A Foundation for Efficiently Adaptable VLA Models",
     "url": "https://arxiv.org/abs/2609.38164",
     "type": "paper",
     "publisher": "Microsoft Research",
     "date": "2026-09",
     "accessed": "2026-10-11"
    },
    "s29": {
     "title": "Continual World: A Robotic Benchmark For Continual Reinforcement Learning",
     "url": "https://arxiv.org/abs/2105.10919",
     "type": "paper",
     "publisher": "NeurIPS 2021",
     "date": "2021-05",
     "accessed": "2026-10-11"
    },
    "s30": {
     "title": "Benchmarking Massively Parallelized Multi-Task Reinforcement Learning for Robotics Tasks (MTBench)",
     "url": "https://arxiv.org/abs/2507.23172",
     "type": "paper",
     "publisher": "RLC 2025",
     "date": "2025-07",
     "accessed": "2026-10-11"
    },
    "s31": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies",
     "url": "https://arxiv.org/abs/2512.16881",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (2026 audit)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Meta-World documentation site (HTTP Last-Modified 2026-09-12)",
     "url": "https://metaworld.farama.org/",
     "type": "site",
     "publisher": "Farama Foundation",
     "date": "2026-09",
     "accessed": "2026-10-11"
    },
    "s34": {
     "title": "X2Real Technical Report (related work on simulation benchmarks)",
     "url": "https://arxiv.org/abs/2609.27449",
     "type": "paper",
     "publisher": "X Square Robot",
     "date": "2026-09",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from the checked basic entry and primary sources. sim_to_real changed from 'none-found' to 'demonstrated'; added the VLA tier-average score series, RL protocol results and version history. Most Meta-World sources were opened on 2026-10-11 (local time) and carry that date."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "motion-turing-test",
   "name": "Motion Turing Test",
   "full_name": "Motion Turing Test (HHMotion)",
   "aliases": [
    "MTT",
    "HHMotion",
    "Human-Humanoid Motion dataset",
    "PTR-Net baseline"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Produces a human-likeness score for humanoid robot motion (real and simulated robots), a capability listed in the taxonomy. Its second task scores automatic raters, not robots.",
   "summary": {
    "text": "People rate how human-like humanoid robot motions look (0-5), using pose-only replays of 1,000 human and robot clips.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Xiamen University (several labs), OPPO Research Institute, ShanghaiTech University.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Affiliations listed in the arXiv HTML; the project site's licence page names Xiamen University's spAital Sensing and Computing Lab and ShanghaiTech as dataset licensor."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Also listed at https://cvpr.thecvf.com/virtual/2026/poster/37546. arXiv comments field does not name the venue."
    },
    "first_release": {
     "value": "2026-03 (arXiv v1 submitted 2026-03-06; project page dated 2026-03-13).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "2026-09-15: commit 'Add PtrNet implementation' in GitHub repo LMZZZZZZZZ/MotionTuringTest (commit author 'Mingzhe Li'). No newer arXiv version.",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Repo is on a personal account, not linked from the paper, CVF page or project page; treated as probably the first author's. Its data README points to the project page for downloads."
    },
    "version": {
     "value": "arXiv v1 only; CVPR 2026 open-access version.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "human-likeness",
      "locomotion"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Full list of 15 categories is only in figures; text names about 12."
    },
    "embodiment": {
     "value": [
      "humanoid"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Complete robot list is in a supplement figure (image), not machine-readable."
    },
    "scene": {
     "value": [],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "1,000 5-second clips: 500 humanoid (257 real-robot, 243 simulated) + 500 human (365 from 10 subjects, 135 from YouTube); 15 categories; 11 humanoid models.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Annotation: 30 annotators, 0-5 Likert, >500 hours; 5 removed by an inter-annotator consistency filter (25 kept).",
       "level": "verified",
       "sources": [
        "s6"
       ],
       "note": "Abstract says 30 annotators; supplement Tab. S2 shows 25 retained."
      }
     ]
    },
    "scoring": {
     "value": [
      "human-rating"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (No leaderboard; results only in the paper.)"
    },
    "license_code": {
     "value": "MIT (LICENSE in PtrNet/ folder, copyright 'Heath000').",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Repo not linked from paper or project page, so its official status is inferred."
    },
    "license_data": {
     "value": "Unknown: HHMotion not released. Site-wide LiDAR Human dataset licence is custom, non-commercial research only, no redistribution.",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Site licence applies 'in the absence of specific indication'; whether it will cover HHMotion is not stated. Paper text itself is CC BY-NC-ND 4.0 on arXiv."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "Real robots (Not applicable in the policy sense; perceived sim-vs-real gap reported (simulated robot motion rated more human-like than real-robot motion).)",
     "note": "No policy is run. Humans rate recorded motion of real robots (257 clips) and simulated robots (243 clips). The paper reports simulated humanoid clips score higher than real-robot clips, i.e. a sim-vs-real gap in perceived human-likeness, not a policy-level correlation."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Towards Motion Turing Test: Evaluating Human-Likeness in Humanoid Robots",
     "url": "https://arxiv.org/abs/2603.06181",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-03"
    },
    "s2": {
     "title": "Towards Motion Turing Test: Evaluating Human-Likeness in Humanoid Robots (full text)",
     "url": "https://arxiv.org/html/2603.06181v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-03"
    },
    "s3": {
     "title": "CVPR 2026 Open Access Repository",
     "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Li_Towards_Motion_Turing_Test_Evaluating_Human-Likeness_in_Humanoid_Robots_CVPR_2026_paper.html",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "LMZZZZZZZZ/MotionTuringTest on GitHub (repository)",
     "url": "https://github.com/LMZZZZZZZZ/MotionTuringTest",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "Towards Motion Turing Test: Evaluating Human-Likeness in Humanoid Robots",
     "url": "http://www.lidarhumanmotion.net/mtt/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "https://openaccess.thecvf.com/content/CVPR2026/supplemental/Li_Towards_Motion_Turing_CVPR_2026_supplemental.pdf",
     "url": "https://openaccess.thecvf.com/content/CVPR2026/supplemental/Li_Towards_Motion_Turing_CVPR_2026_supplemental.pdf",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "License",
     "url": "http://www.lidarhumanmotion.net/license/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2603.06181/citations",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "multion",
   "name": "MultiON",
   "aliases": [
    "Multi-Object Navigation",
    "MultiON Challenge",
    "m-ON (1-ON, 2-ON, 3-ON)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores simulated navigation agents that must find a sequence of objects in 3D homes; fixed episodes, metrics and an organiser-run leaderboard.",
   "summary": {
    "text": "Simulation benchmark where an agent must find several target objects in a set order in 3D homes.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "IIT Kanpur, UIUC, Simon Fraser University (paper); challenge hosted by SFU (3dlg-hcvc, EvalAI team 'Multion Team', sfu-evalai)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Challenge host from https://eval.ai/api/challenges/challenge/2276/"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "India (IIT Kanpur) + USA (UIUC) + Canada (SFU)."
    },
    "first_release": {
     "value": "2020-12 (arXiv v1 2020-12-07)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "MultiON Challenge 2024 at CVPR 2024 Embodied AI Workshop (start 2024-04-19, deadline 2024-06-03); EvalAI challenge 2276 '2024-2025' ran to 2025-08-01; challenge repo last commit 2024-05-30",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "EvalAI dates: https://eval.ai/api/challenges/challenge/2276/"
    },
    "version": {
     "value": "2024 task: 3 targets per episode described by language (e.g. 'Find the mantel clock on the chest of drawers'), goal set not known a priori, HSSD scenes",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "navigation",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "mobile-base"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "2024 sensors from http://multion-challenge.cs.sfu.ca"
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "Per m-ON dataset: 50,000 train episodes per scene; 12,500 val and 12,500 test episodes per scene; MP3D standard splits; inter-goal distance 2-20 m",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "progress",
      "path-efficiency",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Definitions also in https://arxiv.org/pdf/2012.03912v1"
    },
    "evaluator": {
     "value": "organiser-run",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "The organisers (EvalAI minival / test-standard / test-challenge phases)"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "Official (EvalAI). 2024 test-standard: 2 entries)"
    },
    "top_score": {
     "value": "2024 test-standard: IntelliGO Labs Progress 0.058, PPL 0.032, Success 0.008 (2024-06-13)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "license_code": {
     "value": "MIT (3dlg-hcvc/multion-challenge); original saimwani/multiON repo has no licence per GitHub",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Original repo: https://api.github.com/repos/saimwani/multiON"
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked at challenge site Terms and Conditions and repo READMEs."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No real-robot evaluation in the paper or challenge pages; none found in a web search."
    },
    "citations": {
     "value": 158,
     "display": "158 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Last edition 2024; EvalAI closed 2025-08-01; no 2025/2026 edition listed on the challenge site."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "MultiON: Benchmarking Semantic Map Memory using Multi-Object Navigation",
     "url": "https://arxiv.org/abs/2012.03912",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2020-12"
    },
    "s2": {
     "title": "MultiON: Benchmarking Semantic Map Memory using Multi-Object Navigation",
     "url": "https://arxiv.org/pdf/2012.03912v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2020-12"
    },
    "s3": {
     "title": "MultiON: Benchmarking Semantic Map Memory using Multi-Object Navigation",
     "url": "https://shivanshpatel35.github.io/multi-ON/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "MultiON EAI Challenge CVPR 2024",
     "url": "http://multion-challenge.cs.sfu.ca",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "https://eval.ai/api/challenges/challenge/2002/",
     "url": "https://eval.ai/api/challenges/challenge/2002/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "EvalAI: Evaluating state of the art in AI",
     "url": "https://eval.ai/web/challenges/challenge-page/2276/leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "https://eval.ai/api/jobs/challenge_phase_split/5633/leaderboard/",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/5633/leaderboard/",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "3dlg-hcvc/multion-challenge on GitHub (repository)",
     "url": "https://api.github.com/repos/3dlg-hcvc/multion-challenge",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/ARXIV:2012.03912?fields=citationCount",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "mvpbench",
   "name": "MVPBench",
   "full_name": "MVPBench (Minimal Video Pairs)",
   "aliases": [
    "MVP",
    "Minimal Video Pairs",
    "MVP-mini",
    "mvp_mini"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "General video-QA benchmark on physical understanding. One of four subsets is robot-arm video (Language Table): 25,796 of 54,828 full-split rows. Not built only for robots, so 'borderline'.",
   "summary": {
    "text": "Video question set on physical understanding; each question comes as a near-identical video pair with opposite answers.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "FAIR at Meta (all authors); first author also Mila and McGill University, work done during internship",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-06 (arXiv v1 2025-06-11)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "HF dataset object created 2025-03-24 (API), announced with the blog 2025-06-11."
    },
    "latest_update": {
     "value": "OpenReview: 'Accepted by TMLR', publication date 2025-12-01. Last repo commit 2025-09-22 (download script fix).",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "OpenReview API search; Semantic Scholar venue also 'Trans. Mach. Learn. Res.'; commit date from GitHub API."
    },
    "version": {
     "value": "arXiv v1 only; TMLR 2025 publication",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Physical world understanding and spatio-temporal reasoning in video-language models."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "55K examples from 9 sources; human 92.9%, best open-source model 40.2%, random 25%.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "HF card: full 54,828 rows; mini 18,416 rows (5,724 + 8,680 + 2,000 + 2,012).",
       "level": "verified",
       "sources": [
        "s5"
       ],
       "note": "Sums are ours."
      }
     ]
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "V-JEPA 2 (8B VLM setup) 44.5 paired accuracy, reported by Meta in the V-JEPA 2 paper.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Self-reported by the same lab."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "display": "Official (: Meta 'Physical Reasoning from Video' HF Space, tracking the mini split. Showed 'Runtime error' on 2026-10-10.)"
    },
    "license_code": {
     "value": "CC-BY-NC-4.0 (repo LICENSE file)",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "GitHub reports 'Other'; README says benchmark released under this LICENSE."
    },
    "license_data": {
     "value": "HF card metadata: Apache-2.0. Videos are not hosted 'for legal reasons'; users download them from the nine original sources and accept their licences.",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Conflicts with repo LICENSE (CC-BY-NC-4.0). Flagged."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Not built to predict robot performance. Searched paper, repo, HF card."
    },
    "citations": {
     "value": 30,
     "display": "30 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 39,
     "display": "39",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "used_by": {
     "value": "Meta V-JEPA 2 paper; NVIDIA Cosmos 3 report (video/physical reasoning evaluation).",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "V-JEPA 2: https://arxiv.org/html/2506.09985v1"
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Mini split described as '9k examples' (readme, hf card) and '~5k items' (leaderboard text)",
     "text": "Likely different units (rows vs pairs); not resolved.",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "shortcut",
     "title": "The benchmark is designed against shortcut solutions; authors show language-only, single-f",
     "text": "Design feature, recorded for context.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs (full text)",
     "url": "https://arxiv.org/html/2506.09987v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s2": {
     "title": "A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs",
     "url": "https://arxiv.org/abs/2506.09987",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s3": {
     "title": "https://api2.openreview.net/notes/search?term=Minimal%20Video%20Pairs%20shortcut-aware",
     "url": "https://api2.openreview.net/notes/search?term=Minimal%20Video%20Pairs%20shortcut-aware",
     "type": "index",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "facebookresearch/minimal_video_pairs on GitHub (repository)",
     "url": "https://github.com/facebookresearch/minimal_video_pairs",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "facebook/minimal_video_pairs on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/facebook/minimal_video_pairs",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "facebook/physical_reasoning_leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/facebook/physical_reasoning_leaderboard/blob/main/content.py",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s7": {
     "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning (full text)",
     "url": "https://arxiv.org/html/2506.09985v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-06"
    },
    "s8": {
     "title": "facebook/physical_reasoning_leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/facebook/physical_reasoning_leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s9": {
     "title": "facebookresearch/minimal_video_pairs on GitHub (blob)",
     "url": "https://github.com/facebookresearch/minimal_video_pairs/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s11": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2606.02800v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "open-x-embodiment",
   "name": "Open X-Embodiment",
   "full_name": "Open X-Embodiment: Robotic Learning Datasets and RT-X Models",
   "aliases": [
    "OXE",
    "Open X-Embodiment Dataset",
    "OpenX",
    "Open X-Embodiment Repository",
    "RT-X",
    "RT-1-X",
    "RT-2-X"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "A pooled training dataset with no test of its own. It is in scope because its robot setups and data are the basis of common evaluations: the paper's RT-X real-robot study (3,600 trials on 6 robots), SimplerEnv's simulated Google Robot and WidowX/Bridge setups, offline action-error scoring on held-out episodes, and world-model evaluators trained on it (WorldGym).",
   "summary": {
    "text": "Open X-Embodiment pools robot datasets from many labs into one data format. The paper also trains two models on it, RT-1-X and RT-2-X, and tests them in 3,600 real-robot trials on 6 robots. It has no fixed test of its own; its Google Robot and WidowX setups are the basis of later evaluations such as SimplerEnv.",
    "short": "Open X-Embodiment (OXE) pools robot data from dozens of labs into one format, and model builders use it as training data. It has no test of its own.",
    "sources": [
     "s2",
     "s4",
     "s22"
    ]
   },
   "facts": {
    "kind": {
     "value": "dataset",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The paper and README present a dataset (the 'Open X-Embodiment Repository') and model checkpoints. There is no fixed test set, scoring rule or leaderboard."
    },
    "kind_secondary": {
     "value": [
      "study"
     ],
     "display": "Also two model families (RT-1-X, RT-2-X) and a one-off real-robot evaluation in the paper",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The taxonomy has no value for a model."
    },
    "publishers": {
     "value": [
      "Open X-Embodiment Collaboration",
      "Google DeepMind"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s5",
      "s12"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Open X-Embodiment Collaboration",
       "display": "293 authors on arXiv v9 (Crossref lists 279). The PDF author footnote has 46 numbered affiliations; UT Austin appears twice (entries 30 and 45).",
       "level": "verified",
       "sources": [
        "s2",
        "s3",
        "s13"
       ]
      },
      {
       "value": "Google DeepMind",
       "display": "Owns the GitHub repository and hosts the data bucket. README copyright: DeepMind Technologies Limited. Announced the release on its blog on 2023-10-03.",
       "level": "verified",
       "sources": [
        "s5",
        "s12",
        "s9"
       ]
      },
      {
       "value": "Institution count differs by source",
       "display": "21 institutions (abstract), 33 academic labs (Google DeepMind blog), 34 labs (paper Section III-A), 46 affiliation entries (PDF footnote).",
       "level": "verified",
       "sources": [
        "s1",
        "s2",
        "s3",
        "s12"
       ],
       "note": "The numbers count different things (institutions, labs, affiliation lines). Report a range."
      }
     ],
     "note": "Contributors can still enrol datasets through a form linked on the project site."
    },
    "builder_type": {
     "value": "consortium",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "From the author list: universities, AI labs (Google DeepMind, Meta AI, NVIDIA, Microsoft Research) and companies (Intrinsic, Flexiv, IO-AI TECH) under one collaboration byline."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Affiliations in North America, Europe (e.g. ETH Zurich, Freiburg, DLR, Edinburgh, Imperial College), Asia (Tokyo, KAIST, Shanghai Jiao Tong, Tsinghua) and Australia (QUT)."
    },
    "first_release": {
     "value": "2023-10",
     "display": "Announced 2023-10-03 on the Google DeepMind blog. arXiv v1 2023-10-13. Published at ICRA 2024.",
     "level": "verified",
     "sources": [
      "s12",
      "s1",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Bucket folder markers for v1.0 datasets date from 2023-09-27 to 2023-10-05. The GitHub repo was created 2023-10-20.",
     "short": "October 2023, published at ICRA 2024"
    },
    "published_at": {
     "value": "ICRA 2024",
     "display": "IEEE ICRA 2024, pages 6892-6903, DOI 10.1109/ICRA57147.2024.10611477",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Crossref record deposited by IEEE (published 2024-05-13). The arXiv page lists no venue."
    },
    "latest_update": {
     "value": "2026-04",
     "display": "2026-04-27: a UR5e dataset (robo_ai_u_r5e, 443 episodes) appeared in the data bucket. It is not in the spreadsheet or README. Last repo commit 2025-11-05 (an install fix). Last paper version: arXiv v9, 2025-05-14.",
     "level": "verified",
     "sources": [
      "s43",
      "s9",
      "s7",
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Dates are bucket object timestamps and commit dates. We found no announcement for the 2026 addition.",
     "short": "April 2026. One dataset was added but not listed."
    },
    "version": {
     "value": "v1.0 + v1.1",
     "display": "The official spreadsheet groups 72 datasets into v1.0 (60) and v1.1 (12). Each dataset has its own TFDS version folder (e.g. 0.1.0). The repo has no tags or releases.",
     "level": "verified",
     "sources": [
      "s8",
      "s9",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v1.0",
       "display": "60 datasets, 1,403,481 episodes, 4,435.41 GB",
       "level": "inferred",
       "sources": [
        "s8"
       ],
       "note": "Summed by us from the spreadsheet rows (CSV export, 2026-10-10)."
      },
      {
       "value": "v1.1",
       "display": "12 datasets (DROID, ConqHose, DobbE, FMB, IO-AI Office PicknPlace, MimicPlay, MobileALOHA, RoboSet, TidyBot, VIMA, SPOC, Plex RoboSuite), 1,015,712 episodes, 4,529.53 GB. Their bucket folders were created between 2024-03-21 and 2024-08-05.",
       "level": "inferred",
       "sources": [
        "s8",
        "s9"
       ],
       "note": "Counts summed from spreadsheet rows; dates from bucket folder timestamps. The spreadsheet gives no release date."
      },
      {
       "value": "Unlisted additions",
       "display": "bridge_data_msr (822 WidowX episodes from Microsoft Research, folder created 2024-04-11) and robo_ai_u_r5e (443 UR5e episodes from Satakunta University of Applied Sciences, 2026-04-27) are in the bucket but not in the spreadsheet.",
       "level": "verified",
       "sources": [
        "s9",
        "s42",
        "s43",
        "s8"
       ]
      }
     ],
     "short": "v1.0 and v1.1. The repository has no tagged releases."
    },
    "status": {
     "value": "maintained",
     "display": "Low activity. One unlisted dataset was added to the bucket in April 2026. The spreadsheet and README do not list it. Reported data defects remain open.",
     "level": "inferred",
     "sources": [
      "s7",
      "s9",
      "s6",
      "s46"
     ],
     "checked": "2026-10-10",
     "note": "56 open issues and pull requests (GitHub API). The last merged change is the 2025-11-05 install fix. Use as pretraining data is high (facts.used_by).",
     "short": "Maintained, with low activity"
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The paper focuses on robotic manipulation with language-conditioned policies. The pool also holds 3 wheeled-robot navigation datasets and 1 quadruped dataset (spreadsheet), which get no capability tag here."
    },
    "generalisation": {
     "value": [
      "new-task",
      "object-instance",
      "visual",
      "scene-layout"
     ],
     "display": "In the RT-X study: skills seen only in another robot's data, and unseen objects, backgrounds and environments. The small-data domains were tested in distribution.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Describes the paper's evaluation, Table II ('Emergent Skills' and 'RT-2 Generalization' columns). The dataset itself defines no test conditions.",
     "short": "New skills, objects and scenes in the RT-X study"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Describes the RT-X evaluation. The data is also used offline (action prediction error) and through simulated or learned replicas (see validity)."
    },
    "simulator": {
     "value": "none",
     "display": "None for the RT-X tests, which ran on real robots. Five pooled datasets are simulation data (facts.demonstrations).",
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "cross-embodiment"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "22 embodiments (paper); 27 robot names in the spreadsheet",
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RT-X test robots (6)",
       "display": "WidowX (Stanford IRIS, Berkeley RAIL), Google Robot (Google), Franka (Freiburg; Berkeley cable routing), Jaco 2 (USC), Hello Stretch (NYU), UR5 (Berkeley AUTOLab)",
       "level": "inferred",
       "sources": [
        "s2",
        "s4",
        "s8"
       ],
       "note": "Matched by us from the paper's list of evaluation datasets, the spreadsheet's robot column and the project site's list of evaluating labs."
      },
      {
       "value": "Spreadsheet morphology counts",
       "display": "54 single arm, 10 mobile manipulator, 3 wheeled robot, 2 bimanual, 1 quadruped, 1 human, 1 mixed robot and human",
       "level": "inferred",
       "sources": [
        "s8"
       ],
       "note": "Counted by us from the 'Robot Morphology' column of 72 rows."
      }
     ],
     "short": "22 robot types"
    },
    "scene": {
     "value": [
      "tabletop",
      "kitchen",
      "home",
      "mixed"
     ],
     "display": "Scene tags across 72 datasets: tabletop 56, kitchen (including toy kitchens) 24, other household 11, hallways 8, outdoors 1",
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Counted by us from the spreadsheet; one dataset can carry several tags."
    },
    "tasks": {
     "value": 160266,
     "display": "160,266 tasks grouped into 527 skills (paper abstract). The blog says more than 500 skills and 150,000 tasks.",
     "level": "verified",
     "sources": [
      "s1",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Skills and objects were extracted from the language annotations with the PaLM language model (Section III-B), so datasets without annotations do not contribute. Not comparable with task counts of fixed benchmarks.",
     "short": "160,266 tasks in 527 skills"
    },
    "scenes": {
     "value": 300,
     "display": "About 300 scenes in total, by the DROID authors' count",
     "level": "reported",
     "sources": [
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "OXE's own documents give no total. The paper plots distinct scenes per robot type (Fig. 1b) without a total. The DROID paper says OXE spans 'approximately 300 scenes total'."
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No object count in the paper, project site, README or spreadsheet. Fig. 1(e) of the paper shows common object categories without a total."
    },
    "demonstrations": {
     "value": 2419193,
     "display": "2,419,193 episodes in 72 datasets (spreadsheet header, 2026-10-10). The paper says 1M+ trajectories from 60 datasets.",
     "level": "verified",
     "sources": [
      "s8",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "1M+ (paper)",
       "display": "'1M+ real robot trajectories' from 60 pooled datasets (Section III-A)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "924,553 simulated episodes (38.2%)",
       "display": "VIMA 660,103 and SPOC 233,000 (both simulation per their own papers), ManiSkill 30,000, USC Cloth Sim 1,000 and Plex RoboSuite 450. They make up 88.0% of the v1.1 batch.",
       "level": "inferred",
       "sources": [
        "s8",
        "s31",
        "s32",
        "s33"
       ],
       "note": "VIMA and SPOC confirmed as simulation at their arXiv abstracts. ManiSkill2 is a simulator. USC Cloth Sim and Plex RoboSuite identified from names and descriptions. Shares computed by us."
      },
      {
       "value": "Four datasets hold 79.2%",
       "display": "VIMA 27.3%, QT-Opt 24.0%, Language Table 18.3% and SPOC 9.6% of all listed episodes",
       "level": "inferred",
       "sources": [
        "s8"
       ],
       "note": "Computed by us from the spreadsheet rows."
      },
      {
       "value": "Human-body data",
       "display": "IO-AI Office PicknPlace (3,847 episodes) is recorded on human hands with motion capture. RoboVQA (61,153) mixes robot and human episodes.",
       "level": "verified",
       "sources": [
        "s8"
       ]
      },
      {
       "value": "27 of 72 without language",
       "display": "26 datasets list 'None' for language annotations and 1 states it has none. 15 were collected by scripted policies and 13 by expert policies. 26 are flagged as containing suboptimal data.",
       "level": "inferred",
       "sources": [
        "s8"
       ],
       "note": "Counted by us from spreadsheet columns."
      },
      {
       "value": "Subsets used for training",
       "display": "RT-X robotics mixture: 12 datasets from 9 manipulators (paper). Octo: 800k episodes from 25 datasets. OpenVLA: 970k episodes. Octo and OpenVLA put the RT-X subset at 350K episodes; the OXE paper gives no count.",
       "level": "verified",
       "sources": [
        "s2",
        "s24",
        "s25"
       ]
      }
     ],
     "short": "2.4 million episodes in 72 datasets"
    },
    "scale": {
     "value": "8,964.94 GB",
     "display": "8,964.94 GB current download size for 72 datasets (spreadsheet header): v1.0 4,435.41 GB, v1.1 4,529.53 GB",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The per-version split is summed by us from the rows.",
     "short": "About 9 TB"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "RT-X results are success rates on tasks defined by each lab. Offline use scores action prediction error (MSE) on held-out episodes, which has no taxonomy value."
    },
    "metric_detail": {
     "value": "RT-X study: success rate against per-dataset baselines",
     "display": "Success rate per lab-defined task. Small-data domains (5 datasets): RT-1-X against each dataset's 'Original Method' and an RT-1 trained on that dataset alone. Large-data domains: Bridge (WidowX) and RT-1 (Google Robot). Emergent-skill and generalisation tests on the Google Robot.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RT-1-X, small-data domains",
       "display": "Mean success 50% higher (relative) than the Original Method or RT-1; better than the Original Method on 4 of 5 datasets",
       "level": "verified",
       "sources": [
        "s2",
        "s12"
       ],
       "note": "Per-domain numbers are in Fig. 3 (image), not in the text."
      },
      {
       "value": "Large-data domains (Table I)",
       "display": "Bridge at Stanford IRIS / Berkeley RAIL: LCBC 13% / 13%, RT-1 40% / 30%, RT-1-X 27% / 27%, RT-2-X (55B) 50% / 30%. Google Robot: RT-1 92%, RT-1-X 73%, RT-2-X 91%.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "The Table I caption says RT-1-X performs worse than the Original Method and RT-1, but on Bridge RT-1-X (27%) beat the Original Method (13%)."
      },
      {
       "value": "RT-2-X emergent skills (Table II)",
       "display": "RT-2-X (55B) 75.8% vs RT-2 27.3% (about 3x); 42.8% when Bridge data is removed from training. RT-2 generalisation test: 61% vs 62%.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "Success rate on each lab's own tasks"
    },
    "trials": {
     "value": 3600,
     "display": "3,600 real-robot trials across 6 robots in the RT-X study (total). Per-lab counts are not given in the text.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Later real tests in OXE setups",
       "display": "OpenVLA: Google Robot 12 tasks x 5 trials (60 rollouts per policy); WidowX 17 tasks x 10 trials (170). Octo: 2 language tasks x 10 trials per robot.",
       "level": "verified",
       "sources": [
        "s25",
        "s24"
       ]
      }
     ],
     "short": "3,600 real-robot trials in total"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "The OXE paper text reports no error bars or intervals. Some later real-robot reports in OXE setups do: OpenVLA gives standard errors."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Each lab ran RT-X trials on its own setup; later papers run their own trials. No organiser runs submissions."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s5",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the project site, README or spreadsheet."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s5",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "README: all software under Apache 2.0. The GitHub licence API reports Apache-2.0 for the LICENSE file."
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "display": "Blanket notice: 'All other materials' are under CC BY 4.0 (README, and a README.pdf in the data bucket). No per-dataset licences are listed.",
     "level": "verified",
     "sources": [
      "s5",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "The README also asks users to cite each contributed dataset. Whether the blanket notice replaces the original terms of contributed data is not stated. Not legal advice.",
     "items": [
      {
       "value": "No licence column or files",
       "display": "The spreadsheet has no licence column. In the 101 dataset folders of gs://gresearch/robotics, the only licence file we found is DROID's (robotics/droid/1.0.0/CC-BY-4.0).",
       "level": "verified",
       "sources": [
        "s8",
        "s9"
       ],
       "note": "Searched object names for licence-like strings in every folder (2026-10-10)."
      },
      {
       "value": "CC BY 4.0 at source",
       "display": "Stated at the source for BridgeData V2, DROID, RoboNet, VIMA (VimaBench README), RoboVQA and CLVR Jaco Play",
       "level": "verified",
       "sources": [
        "s47",
        "s9",
        "s34",
        "s35",
        "s36",
        "s37"
       ]
      },
      {
       "value": "CC BY-NC 4.0 at source (ManiSkill2 assets)",
       "display": "ManiSkill2 (30,000 simulated episodes in OXE) states its assets are under CC BY-NC 4.0 (README at tag v0.5.3).",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "No data licence found at source",
       "display": "Language Table and SPOC repos state code licences (Apache-2.0) only; we found no separate data licence statement in their READMEs.",
       "level": "verified",
       "sources": [
        "s38",
        "s39"
       ]
      },
      {
       "value": "cc-by-4.0 (third-party mirror)",
       "display": "Hugging Face mirror jxu124/OpenX-Embodiment, 19,997 downloads (Hub 'downloads' field)",
       "level": "reported",
       "sources": [
        "s45"
       ],
       "note": "Label read on the third-party mirror card; it does not decide OXE terms."
      }
     ],
     "short": "One CC BY 4.0 notice covers all datasets. The terms at the original sources vary."
    },
    "license_assets": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s33",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Only the simulated datasets involve 3D assets. ManiSkill2 states its assets are CC BY-NC 4.0; OXE ships rendered observations, not the asset files, and no source says how the asset terms apply to renders. VIMA and SPOC asset terms were not checked."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s5",
      "s9",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Public bucket gs://gresearch/robotics via TFDS or gsutil; no registration. The README's fallback download bucket (gs://gdm-robotics-open-x-embodiment) has data in only 35 of its 72 dataset folders; the other 37, including bridge and fractal20220817_data, hold only folder markers (our listing, 2026-10-10). See issues.i4.",
     "short": "Open. The data is in a public storage bucket."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s5",
      "s8",
      "s33",
      "s38",
      "s39"
     ],
     "checked": "2026-10-10",
     "note": "Code (Apache-2.0) and the blanket CC BY 4.0 notice both allow commercial use with attribution. But the pool re-hosts data from many labs; one component states non-commercial terms for its assets (ManiSkill2) and several state no data licence at all. So the whole pool is unclear; individual datasets may be clear. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "display": "Scores in the OXE paper come from real robots. Evaluations built on OXE data have been checked against real robots (see validity).",
     "level": "inferred",
     "sources": [
      "s2",
      "s22",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Three paired comparisons use OXE data or setups. SimplerEnv's simulated Google Robot setup tracked real results for 6 policies (Pearson r 0.924, MMRV 0.056). Action prediction error on Google Robot episodes did not (r 0.308, MMRV 0.375). WorldGym, a world model trained on 9 OXE datasets, matched OpenVLA's real Bridge trials at r 0.78 per task for 3 policies. Studies tied to a single component are in the BridgeData V2 and DROID records (SimplerEnv WidowX, AutoEval, PolaRiS, REALM and others). All pairings use 3 to 6 policies; none reports a confidence interval on r.",
     "short": "Its scores come from real robots. Tests built on OXE data have been checked against real robots."
    },
    "real_reproducibility": {
     "value": "multi-site-measured",
     "display": "Small: the RT-X study ran the same 4 models in the Bridge setup at two labs (Stanford IRIS, Berkeley RAIL).",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "LCBC 13% / 13%, RT-1 40% / 30%, RT-1-X 27% / 27%, RT-2-X 50% / 30%. Pearson r between the two labs is 0.89 by our calculation (4 points). Every other domain was tested at one lab."
    },
    "citations": {
     "value": 1229,
     "display": "1,229 (Semantic Scholar; 83 influential)",
     "level": "verified",
     "sources": [
      "s14"
     ],
     "checked": "2026-10-10",
     "short": "1,229"
    },
    "github_stars": {
     "value": 2059,
     "display": "2,059 stars, 129 forks (google-deepmind/open_x_embodiment)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "2,059"
    },
    "used_by": {
     "value": "Pretraining data in at least 8 model reports we checked",
     "level": "verified",
     "sources": [
      "s2",
      "s24",
      "s25",
      "s26",
      "s27",
      "s28",
      "s29",
      "s41"
     ],
     "checked": "2026-10-10",
     "note": "A lower bound from the reports we opened, not a census.",
     "items": [
      {
       "value": "RT-1-X, RT-2-X",
       "display": "Google DeepMind and partners, 2023-10. The paper's own models.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Octo",
       "display": "2024-05. 800k episodes from 25 OXE datasets; zero-shot tests on robots from the pretraining data.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "OpenVLA",
       "display": "2024-06. 970k OXE episodes; real tests on WidowX (Bridge) and Google Robot.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "SpatialVLA",
       "display": "2025-01. An OXE subset plus RH20T for pretraining.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "FAST / π0-FAST",
       "display": "Physical Intelligence, 2025-01. OpenX, DROID and BridgeV2 make up 3.8%, 11.2% and 5.0% of the tokenizer's training mix.",
       "level": "verified",
       "sources": [
        "s41"
       ]
      },
      {
       "value": "GR00T N1",
       "display": "NVIDIA, 2025-03. Pretraining includes the OXE subsets RT-1, Bridge-v2, Language Table, DROID, MUTEX, RoboSet and Plex.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "π0, π0.5",
       "display": "Physical Intelligence, 2024-10 and 2025-04. π0 pretrains on an OXE subset ('OXE Magic Soup'); open-source data is 9.1% of its mixture. π0.5 uses an extended version.",
       "level": "verified",
       "sources": [
        "s26",
        "s27"
       ]
      }
     ],
     "short": "Pretraining data in at least 8 model reports"
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Physical Intelligence",
      "NVIDIA",
      "Microsoft"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s26",
      "s28",
      "s42"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "Built RT-1-X and RT-2-X; hosts the repository and data bucket.",
       "level": "verified",
       "sources": [
        "s2",
        "s12"
       ]
      },
      {
       "value": "Physical Intelligence",
       "display": "π0 and π0.5 pretraining mixtures include OXE.",
       "level": "verified",
       "sources": [
        "s26",
        "s27"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "GR00T N1 pretraining includes seven OXE subsets.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "Microsoft",
       "display": "Microsoft Research contributed Plex RoboSuite (v1.1) and bridge_data_msr (in the bucket, not in the spreadsheet).",
       "level": "verified",
       "sources": [
        "s8",
        "s42",
        "s3"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "SimplerEnv",
      "WorldGym",
      "RobotArena ∞"
     ],
     "display": "Evaluations built from OXE setups or data",
     "level": "verified",
     "sources": [
      "s22",
      "s23",
      "s40"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "SimplerEnv",
       "display": "2024-05. Simulated copies of the Google Robot (RT-1 data) and WidowX (Bridge) setups, for testing OXE-trained policies. Has its own record.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "WorldGym",
       "display": "2025-05. An action-conditioned video world model trained on 9 OXE datasets, used as a test environment.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "RobotArena ∞",
       "display": "2025-10. Simulated scenes rebuilt from videos in Bridge V2 (an OXE subset), DROID and RH20T. No real-robot correlation reported.",
       "level": "verified",
       "sources": [
        "s40"
       ]
      }
     ],
     "short": "3 evaluations built on OXE"
    }
   },
   "validity": [
    {
     "id": "v1",
     "name": "SimplerEnv (simulated Google Robot setup)",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 6 policies (3 RT-1 checkpoints, RT-1-X, RT-2-X and Octo-Base) were scored in a simulated copy of the Google Robot setup and on the real robot, over 3 task groups.",
     "result": "Pearson r = 0.924 and MMRV = 0.056 for Visual Matching. Pearson r = 0.778 and MMRV = 0.143 for Variant Aggregation. MMRV measures how often two rankings disagree.",
     "authors_view": "strong correlation",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s22"
     ],
     "note": "14 of the 16 SimplerEnv authors are OXE authors. Means over 3 tasks (Table I)."
    },
    {
     "id": "v2",
     "name": "Offline action error on Google Robot data (SimplerEnv baseline)",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 6 policies were ranked by their action prediction error on 25 Google Robot (RT-1) training episodes. The ranking was compared with real-robot success over 3 task groups.",
     "result": "Mean over 3 tasks: Pearson r = 0.308, MMRV = 0.375",
     "authors_view": "not a good proxy",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s22"
     ],
     "note": "The authors used training episodes because no validation split of the Google Robot data is public. Table XII gives 0.346 / 0.264 for the drawer task, where Table I gives 0.306 / 0.231."
    },
    {
     "id": "v3",
     "name": "WorldGym (world model trained on OXE)",
     "date": "2025-05",
     "by": "independent",
     "method": "RT-1-X, Octo and OpenVLA were run in a world model (a learned simulator) trained on 9 OXE datasets. Each run started from the first frames of one of OpenVLA's 170 real Bridge trials (17 tasks), and results were compared task by task.",
     "result": "Pearson r = 0.78 for success on each task. The policies' mean scores in the world model differed from their real-robot means by 3.3 points on average.",
     "authors_view": "highly correlate",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s23"
     ],
     "note": "Means real vs world model: RT-1-X 18.5% vs 15.5%, Octo 20.0% vs 23.82%, OpenVLA 70.6% vs 67.4%. A GPT-4o judge scores world-model rollouts."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How often a policy (the robot's control model) will succeed on a real robot.",
     "sub": "Some papers rank policies by action error on OXE episodes, which measures how far predicted actions are from the recorded ones. This ranked policies poorly compared with real-robot success.",
     "basis": [
      "issues.i6"
     ]
    },
    {
     "id": "l2",
     "text": "Whether results from different labs can be compared.",
     "sub": "In the RT-X study, each lab tested the models on its own tasks.",
     "basis": [
      "facts.metric_detail",
      "facts.evaluator",
      "issues.i5"
     ]
    },
    {
     "id": "l3",
     "text": "How much of the data comes from real robots.",
     "sub": "Of the listed episodes, 38% come from simulation.",
     "basis": [
      "facts.demonstrations",
      "issues.i1"
     ]
    }
   ],
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Over a third of listed episodes are simulated",
     "text": "The paper describes '1M+ real robot trajectories'. The spreadsheet now lists 2,419,193 episodes, of which 924,553 (38.2%) come from simulated datasets: VIMA 660,103 and SPOC 233,000 (simulation per their own papers), ManiSkill 30,000, USC Cloth Sim 1,000 and Plex RoboSuite 450. Most of this arrived with v1.1, where simulated data is 88.0% of the batch. Totals that mix v1.0 and v1.1 therefore include a large share of simulated data.",
     "level": "inferred",
     "sources": [
      "s2",
      "s8",
      "s31",
      "s32"
     ],
     "status": "open",
     "note": "Shares computed by us from spreadsheet rows. The spreadsheet does not label any dataset as simulated.",
     "short": "Of the listed episodes, 38% come from simulation. Most of them were added in v1.1."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "Counts differ between the paper, blog and spreadsheet",
     "text": "Institutions: 21 (abstract), 33 academic labs (blog), 34 labs (Section III-A), 46 affiliation entries (PDF). Robots: 22 embodiments (paper) vs 27 robot names (spreadsheet). Datasets: 60 (paper) vs 72 (spreadsheet), with two more in the bucket that the spreadsheet does not list. Episodes: 1M+ (paper) vs 2,419,193 (spreadsheet). Skills and tasks: 527 and 160,266 (paper) vs more than 500 and 150,000 (blog).",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s3",
      "s8",
      "s9",
      "s12"
     ],
     "status": "open",
     "short": "The paper, the blog and the spreadsheet give different numbers of institutions, robots, datasets and episodes."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "The OXE copy of BridgeData V2 is an early partial upload",
     "text": "OXE's 'bridge' dataset has 25,460 train and 3,475 test episodes (28,935 in total), against 60,096 trajectories in BridgeData V2. In December 2023 an OXE author wrote that it was uploaded at an early stage and would be updated. On 2026-10-10 the spreadsheet still lists 25,460 episodes. Octo and OpenVLA train on the Berkeley copy instead; Octo's config notes it is not the official OXE copy. A full copy (bridge_data_v2/0.0.1, 60,063 episodes) has sat in the same bucket since 2024-08-12 without documentation.",
     "level": "verified",
     "sources": [
      "s15",
      "s44",
      "s8",
      "s48",
      "s49",
      "s50"
     ],
     "status": "open",
     "short": "OXE's copy of BridgeData V2 has 28,935 of about 60,000 episodes. Some users, such as Octo and OpenVLA, train on other copies."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Several datasets have reported defects and one download path is half empty",
     "text": "Users report defects in individual datasets on the issue tracker. Roboturk episodes never mark their end; an author confirmed this in 2023-11 and promised a fix, and the issue is still open. Other open reports: Saytap images are all zeros (2024-07), berkeley_rpt lacks two of three camera views (2024-10), and fractal20220817_data contains failed episodes (2024-02). The README's fallback download path, gs://gdm-robotics-open-x-embodiment, has data in only 35 of its 72 dataset folders; the other 37, including bridge and fractal20220817_data, hold only folder markers (our listing, 2026-10-10; issue #104, open since 2025-07).",
     "level": "verified",
     "sources": [
      "s18",
      "s20",
      "s46",
      "s19",
      "s10"
     ],
     "status": "open",
     "note": "The Roboturk defect is confirmed by an author and the empty folders by our own listing. The Saytap, berkeley_rpt and fractal reports are user reports we did not reproduce. The main bucket (gs://gresearch/robotics) holds the data.",
     "short": "Users have reported defects in several datasets. One download path given in the README has data in only 35 of its 72 dataset folders."
    },
    {
     "id": "i5",
     "type": "inconsistent-reporting",
     "title": "The RT-X results cannot be re-run by others",
     "text": "RT-2-X weights were not released: an author replied in 2023-11 that there was 'no possibility to release' them, and in 2024-03 that there were no plans to release RT-2 code. Each lab evaluated on its own tasks; the text gives no per-lab trial counts or error bars. OpenVLA reports that RT-2-X froze on Bridge because every Bridge demonstration starts with an all-zero action, and that the RT-2-X developers queried the second-most-likely action in the OXE Bridge evaluations. The OXE paper does not mention this. The Table I caption also contradicts its own numbers for RT-1-X on Bridge.",
     "level": "verified",
     "sources": [
      "s16",
      "s17",
      "s2",
      "s25"
     ],
     "status": "open",
     "note": "We searched the v9 HTML and PDF for 'second', 'most likely', 'zero' and similar terms: no match.",
     "short": "The RT-2-X weights were not released, and each lab tested on its own tasks. The paper also does not mention a workaround in how RT-2-X's actions were chosen in the Bridge tests."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Offline action error on OXE data does not predict real success",
     "text": "SimplerEnv calls action error on held-out demonstrations a widely adopted way to select policies. On the Google Robot setup it ranked 6 policies poorly (mean Pearson r 0.308, MMRV 0.375). On Bridge tasks its correlation with real success was negative (r -0.342 to -1.000). AutoEval (Bridge) and PolaRiS (DROID) also report negative correlations; see those records.",
     "level": "verified",
     "sources": [
      "s22"
     ],
     "status": "open",
     "short": "When policies were ranked by action error on held-out OXE data, the ranking matched real-robot results poorly. In some tests the order was reversed."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Licence terms of the pooled datasets are unclear",
     "text": "The README puts all non-software materials under CC BY 4.0 and asks users to cite each contributed dataset. The spreadsheet has no licence column, and the only licence file in the bucket is DROID's. At their own sources, several datasets state CC BY 4.0; ManiSkill2 states CC BY-NC 4.0 for its assets; Language Table and SPOC state only code licences.",
     "level": "inferred",
     "sources": [
      "s5",
      "s8",
      "s9",
      "s33",
      "s38",
      "s39"
     ],
     "status": "open",
     "note": "Our reading of the documents. Not legal advice.",
     "short": "One CC BY 4.0 notice covers all the datasets. Some datasets state different terms at their sources, and some state no data licence."
    },
    {
     "id": "i8",
     "type": "other",
     "title": "Actions are only coarsely aligned across datasets",
     "text": "The paper converts each dataset to a 7-dimensional end-effector action but does not align coordinate frames and allows absolute or relative values, so 'the same action vector may induce very different motions for different robots'. Model builders pick subsets and weights by hand: Octo uses 25 datasets, π0 an 'OXE Magic Soup' subset. OpenVLA filtered Bridge's all-zero first actions and dropped DROID for the last third of training.",
     "level": "verified",
     "sources": [
      "s2",
      "s24",
      "s25",
      "s26"
     ],
     "status": "open",
     "short": "Actions are recorded in different coordinate frames and with different meanings across datasets. Model builders therefore choose subsets of the data by hand."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "OXE is training data and a reference for robot setups. It is not a test. A claim to 'evaluate on OXE' can mean real trials in an OXE lab setup, a simulated copy such as SimplerEnv, or action error on held-out episodes. Only the first two have been checked against real robots, and the third ranked policies poorly in every study we found.",
     "basis": [
      "facts.kind",
      "facts.metric_detail",
      "facts.sim_to_real",
      "issues.i6"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "OXE is training data. When a paper reports a score on OXE, check which test it came from."
    },
    {
     "id": "r2",
     "text": "Treat OXE totals with care. The episode count mixes real and simulated data, the Bridge copy is partial, and licence terms vary by dataset. When a model report says it trained on OXE, check which datasets and versions it used.",
     "basis": [
      "issues.i1",
      "issues.i3",
      "issues.i7",
      "facts.demonstrations"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check which OXE datasets a model actually used."
    },
    {
     "id": "r3",
     "text": "The RT-X study is evidence that training on data from many robots can help a given robot. It is one study, and others cannot repeat it, because RT-2-X is closed, each lab used its own tasks, and one evaluation detail went unreported. Later open models trained on OXE, such as Octo and OpenVLA, give evidence that others can check.",
     "basis": [
      "issues.i5",
      "facts.metric_detail",
      "facts.used_by"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The RT-X results come from one study that others cannot repeat."
    }
   ],
   "searched": [
    {
     "for": "validity / sim_to_real",
     "where": "OXE paper v9 (no sim experiments). SimplerEnv (2405.05941) Tables I, IV, XII. WorldGym (2506.00613 v3). AutoEval (2503.24278), PolaRiS (2512.16881), REALM (2512.19562), 2606.10366, RoboWorld (2607.01060) and Ctrl-World (2510.10125): these concern Bridge or DROID and are recorded there. RobotArena ∞ (2510.23571): no real correlation measured. WorldEval (2505.19017) and dWorldEval (2604.22152): checked, they do not use OXE data. Leads from research/raw/sweeps/recent.json followed.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data (constituent datasets)",
     "where": "OXE README and README.pdf in the bucket; spreadsheet columns; all 101 folders of gs://gresearch/robotics searched for licence files; READMEs or licence APIs of RoboNet, VimaBench, VIMA, RoboVQA, CLVR Jaco Play, Language Table, SPOC, ManiSkill (tags v0.4.2, v0.5.0, v0.5.3), TOTO, MimicPlay, RoboHive, Mobile ALOHA, DobbE, FurnitureBench, FMB. The last six give code licences only; not used as evidence of data terms.",
     "date": "2026-10-10"
    },
    {
     "for": "objects, scenes",
     "where": "Paper v9 text and figures captions, project site, README, spreadsheet columns.",
     "date": "2026-10-10"
    },
    {
     "for": "uncertainty_reported",
     "where": "OXE paper v9 HTML and PDF text: no 'standard error', 'confidence' or error-bar wording.",
     "date": "2026-10-10"
    },
    {
     "for": "latest_update",
     "where": "GitHub commits, tags and releases; arXiv version history; bucket listings of gs://gresearch/robotics and gs://gdm-robotics-open-x-embodiment with object timestamps; spreadsheet.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Open X-Embodiment: Robotic Learning Datasets and RT-X Models (arXiv abstract page, v1-v9 history)",
     "url": "https://arxiv.org/abs/2310.08864",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Open X-Embodiment paper, full text v9",
     "url": "https://arxiv.org/html/2310.08864v9",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Open X-Embodiment paper, PDF v9 (author list and affiliation footnote)",
     "url": "https://arxiv.org/pdf/2310.08864v9",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Open X-Embodiment project site",
     "url": "https://robotics-transformer-x.github.io/",
     "type": "site",
     "publisher": "Open X-Embodiment Collaboration",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "open_x_embodiment README (licence, download, RT-1-X checkpoint)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/blob/main/README.md",
     "type": "repo",
     "publisher": "Google DeepMind",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: google-deepmind/open_x_embodiment (stars, forks, licence, open issues)",
     "url": "https://api.github.com/repos/google-deepmind/open_x_embodiment",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "open_x_embodiment commit history",
     "url": "https://github.com/google-deepmind/open_x_embodiment/commits/main",
     "type": "repo",
     "publisher": "Google DeepMind",
     "date": "2025-11-05",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Open X-Embodiment Dataset Overview spreadsheet (read via CSV export)",
     "url": "https://docs.google.com/spreadsheets/d/1rPBD77tk60AEIGZrGSODwyyzs5FgCU9Uz3h-3_t2A9g/edit#gid=0",
     "type": "site",
     "publisher": "Open X-Embodiment Collaboration",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Google Cloud Storage listing of gs://gresearch/robotics (folders, timestamps, licence file search)",
     "url": "https://storage.googleapis.com/storage/v1/b/gresearch/o?prefix=robotics/&delimiter=/",
     "type": "dataset",
     "publisher": "Google",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "Google Cloud Storage listing of gs://gdm-robotics-open-x-embodiment (README fallback bucket)",
     "url": "https://storage.googleapis.com/storage/v1/b/gdm-robotics-open-x-embodiment/o?delimiter=/",
     "type": "dataset",
     "publisher": "Google DeepMind",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "README.pdf in the data bucket (copyright and licence notice)",
     "url": "https://storage.googleapis.com/gresearch/robotics/open_x_embodiment_and_rt_x_oss/README.pdf",
     "type": "dataset",
     "publisher": "Google DeepMind",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Scaling up learning across many different robot types (blog)",
     "url": "https://deepmind.google/blog/scaling-up-learning-across-many-different-robot-types/",
     "type": "blog",
     "publisher": "Google DeepMind",
     "date": "2023-10-03",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Crossref record for the ICRA 2024 paper",
     "url": "https://api.crossref.org/works/10.1109/ICRA57147.2024.10611477",
     "type": "index",
     "publisher": "Crossref (deposited by IEEE)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "Semantic Scholar API record for arXiv:2310.08864",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2310.08864?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Issue #30: discrepancy in the number of trajectories in the bridge dataset (author reply)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/30",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Issue #24: Will there be RT-2-X weights? (author reply)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/24",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2023-11",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Issue #52: When will the code for RT-2 be available? (author reply)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/52",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Issue #11: Roboturk has no True values under is_last or terminate_episode (author confirmation)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/11",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Issue #104: Only 27 of 55 datasets available from the fallback bucket",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/104",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Issue #85: Saytap dataset images are all zeros",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues/85",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Evaluating Real-World Robot Manipulation Policies in Simulation (SimplerEnv), full text: Tables I, IV, V, XII",
     "url": "https://arxiv.org/html/2405.05941",
     "type": "paper",
     "publisher": "arXiv (CoRL 2024)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "WorldGym: World Model as An Environment for Policy Evaluation (v3; Section 4.1, Appendix implementation details)",
     "url": "https://arxiv.org/html/2506.00613v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Octo: An Open-Source Generalist Robot Policy",
     "url": "https://arxiv.org/html/2405.12213",
     "type": "paper",
     "publisher": "arXiv (RSS 2024)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "OpenVLA: An Open-Source Vision-Language-Action Model (Sections 3.3, 5.1; Appendices A, B, C)",
     "url": "https://arxiv.org/html/2406.09246",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "π0: A Vision-Language-Action Flow Model for General Robot Control",
     "url": "https://arxiv.org/html/2410.24164",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "π0.5: a Vision-Language-Action Model with Open-World Generalization",
     "url": "https://arxiv.org/html/2504.16054",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence)",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots",
     "url": "https://arxiv.org/html/2503.14734",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model",
     "url": "https://arxiv.org/html/2501.15830",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset (v2; OXE scene count and OXE co-training baseline)",
     "url": "https://arxiv.org/html/2403.12945v2",
     "type": "paper",
     "publisher": "arXiv (RSS 2024)",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "VIMA: General Robot Manipulation with Multimodal Prompts (abstract: simulation benchmark, 600K+ expert trajectories)",
     "url": "https://arxiv.org/abs/2210.03094",
     "type": "paper",
     "publisher": "arXiv (ICML 2023)",
     "date": "2022-10",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World (abstract)",
     "url": "https://arxiv.org/abs/2312.02976",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "ManiSkill README at tag v0.5.3 (ManiSkill2; licence section: assets CC BY-NC 4.0)",
     "url": "https://github.com/haosulab/ManiSkill/blob/v0.5.3/README.md",
     "type": "repo",
     "publisher": "Hao Su Lab, UC San Diego",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "RoboNet README (data under CC BY 4.0)",
     "url": "https://github.com/SudeepDasari/RoboNet",
     "type": "repo",
     "publisher": "RoboNet authors",
     "date": "2023-03",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "VimaBench README, licence table (dataset CC BY 4.0)",
     "url": "https://github.com/vimalabs/VimaBench",
     "type": "repo",
     "publisher": "VIMA authors",
     "date": "2023-09",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "RoboVQA README, licence and disclaimer",
     "url": "https://github.com/google-deepmind/robovqa",
     "type": "repo",
     "publisher": "Google DeepMind",
     "date": "2023-12",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "GitHub API: clvrai/clvr_jaco_play_dataset (licence CC-BY-4.0)",
     "url": "https://api.github.com/repos/clvrai/clvr_jaco_play_dataset",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "Language Table repository (Apache-2.0; no separate data licence statement in README)",
     "url": "https://github.com/google-research/language-table",
     "type": "repo",
     "publisher": "Google Research",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "SPOC training repository (LICENSE: Apache 2.0)",
     "url": "https://github.com/allenai/spoc-robot-training",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2024-11",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "RobotArena ∞: Scalable Robot Benchmarking via Real-to-Sim Translation",
     "url": "https://arxiv.org/html/2510.23571",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models (Appendix A, data mixture)",
     "url": "https://arxiv.org/html/2501.09747",
     "type": "paper",
     "publisher": "arXiv (Physical Intelligence)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "bridge_data_msr dataset_info.json (Microsoft Research WidowX data in BridgeData V2 format)",
     "url": "https://storage.googleapis.com/gresearch/robotics/bridge_data_msr/0.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "Google (bucket); Microsoft Research (data)",
     "date": "2024-04-11",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "robo_ai_u_r5e dataset_info.json (UR5e data, Satakunta University of Applied Sciences)",
     "url": "https://storage.googleapis.com/gresearch/robotics/robo_ai_u_r5e/0.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "Google (bucket)",
     "date": "2026-04-27",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "TensorFlow Datasets catalog: bridge (OXE copy, 25,460 train / 3,475 test)",
     "url": "https://www.tensorflow.org/datasets/catalog/bridge",
     "type": "dataset",
     "publisher": "Google (TensorFlow Datasets)",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "Hugging Face: jxu124/OpenX-Embodiment (third-party mirror; licence label and downloads)",
     "url": "https://huggingface.co/datasets/jxu124/OpenX-Embodiment",
     "type": "secondary",
     "publisher": "Third party",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "open_x_embodiment issue tracker (incl. #45 failed fractal episodes, #95 missing berkeley_rpt views)",
     "url": "https://github.com/google-deepmind/open_x_embodiment/issues",
     "type": "repo",
     "publisher": "Google DeepMind (issue tracker)",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "BridgeData V2 project site (licence statement)",
     "url": "https://rail-berkeley.github.io/bridgedata/",
     "type": "site",
     "publisher": "UC Berkeley RAIL",
     "date": "2023-08",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "Octo dataset config (note: 'bridge_dataset' is not the official OXE copy)",
     "url": "https://github.com/octo-models/octo/blob/main/octo/data/oxe/oxe_dataset_configs.py",
     "type": "repo",
     "publisher": "Octo Model Team",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "OpenVLA README (download Berkeley RLDS copy, rename to bridge_orig)",
     "url": "https://github.com/openvla/openvla/blob/main/README.md",
     "type": "repo",
     "publisher": "OpenVLA authors",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "bridge_data_v2/0.0.1 dataset_info.json in gs://gresearch/robotics (60,063 episodes)",
     "url": "https://storage.googleapis.com/gresearch/robotics/bridge_data_v2/0.0.1/dataset_info.json",
     "type": "dataset",
     "publisher": "Google (bucket)",
     "date": "2024-08-12",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the checked basic entry and research/raw/inventory/core-real.json. Re-checked every basic fact. Added: v1.0/v1.1 totals and the simulated share, unlisted bucket additions (2024, 2026), licence check of constituent datasets, bucket audit (only licence file is DROID's; fallback bucket half empty), RT-X result tables, Table I caption contradiction, undisclosed RT-2-X Bridge decoding workaround (per OpenVLA), validity list (SimplerEnv, offline MSE, WorldGym), adoption by 8 model reports. Corrected: blog says 33 labs (a fourth institution count); status rests on a 2026-04 bucket addition."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "pai-bench",
   "name": "PAI-Bench",
   "full_name": "PAI-Bench (Physical AI Bench)",
   "aliases": [
    "Physical AI Bench",
    "PAI-Bench-G",
    "PAI-Bench-C",
    "PAI-Bench-U",
    "PAIBench-G (NVIDIA spelling)",
    "PAI-Bench-Predict / -Transfer / -Reason (earlier track names)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Scores video generators and video-language models on physical-AI video. Robotics is one of six domains (107 of 1,044 PAI-Bench-G prompts); the rest are driving, industry, human, physics, common sense. Taxonomy-v0 puts general video-physics benchmarks not built for robots in 'borderline'. It is the automated benchmark NVIDIA uses for its robot-oriented Cosmos world models.",
   "summary": {
    "text": "Tests video generators and video-language models on real-world physical-AI clips: driving, robots, industry, people.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Georgia Tech; CMU (authors Zhou, Huang, Li, Ramanan, Shi). Acknowledgements thank NVIDIA Research, especially the Cosmos team, for support that led to PAI-Bench.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Repo README repeats the same affiliations and acknowledgement."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Author affiliations are universities; NVIDIA supported but is not listed as an affiliation."
    },
    "first_release": {
     "value": "2025-09",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub repo created 2025-09-09; HF datasets created 2025-09-12/21; leaderboard Space created 2025-10-10. arXiv v1 is 2025-12-01. Dates from GitHub and HF APIs."
    },
    "latest_update": {
     "value": "Leaderboard data commit 2026-08-18 'Add Cosmos3-Edge to generation leaderboard'; code commit 2026-06-23 (depth si-RMSE outlier cap; deterministic DOVER scoring).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Code commits at https://github.com/SHI-Labs/physical-ai-bench/commits/main"
    },
    "version": {
     "value": "No version tags or numbered releases; datasets last modified 2025-12-10.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub repo has no releases; HF lastModified from API."
    },
    "published_at": {
     "value": "CVPR 2026 (README says Oral, news dated 2026-04-09)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "CVF open-access page lists the paper in CVPR 2026 proceedings; OpenReview API also lists venue 'CVPR 2026'. Oral status only from README."
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "No control loop: generated videos are scored by metrics and a VLM judge; understanding track is multiple-choice QA."
    },
    "capability": {
     "value": [
      "world-modeling",
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "G and C test video generation (world modelling); U tests physical common sense and embodied reasoning QA."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Per-domain counts are from NVIDIA's Cosmos 3 report describing PAI-Bench-G; the PAI-Bench paper gives the six domain names."
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2,808 cases total (headline).",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "G: 1,044 video-prompt pairs and 5,636 QA pairs across 6 domains.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      },
      {
       "value": "C: 600 videos, 200 clips each from AgiBot (robotics), OpenDV (driving), Ego-Exo4D (egocentric); 1 original + 5 variant captions per video.",
       "level": "verified",
       "sources": [
        "s1"
       ]
      },
      {
       "value": "U: 604 QA pairs on 426 videos (physical common sense) + 610 QA pairs from 601 videos (embodied reasoning: RoboVQA, RoboFail, BridgeData, AgiBot, HoloAssist, a proprietary AV dataset).",
       "level": "verified",
       "sources": [
        "s1"
       ]
      }
     ]
    },
    "scoring": {
     "value": [
      "composite",
      "auto-judge",
      "fidelity",
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "0.5/0.5 weighting stated explicitly in Cosmos 3 report (https://arxiv.org/html/2606.02800v4) and Cosmos-Predict2.5 report; consistent with paper tables (e.g. source videos 89.8 domain, 78.0 quality, 83.9 overall)."
    },
    "human_agreement": {
     "value": "Pearson r = 0.918 between PAI-Bench-G scores and Elo ratings from a pairwise human study (source videos + 8 models; separate Elo for quality and plausibility).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Measured by the benchmark authors."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Official (: Hugging Face Space, running on 2026-10-10. Data files hold 20 rows (generation), 19 rows (conditional generation), 25 rows (understanding).)",
     "note": "Row counts from the Space's JSON data files."
    },
    "top_score": {
     "value": "Generation: Cosmos3-Super 83.9 overall, Cosmos3-Nano 83.7, 'Source' (real videos) 82.6, Veo-3 82.1. Understanding: Cosmos-Reason2-32B 70.8, GPT-5 69.8.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Note the leaderboard's real-video 'Source' row (82.6) differs from the paper's 83.9."
    },
    "evaluator": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "both: self-reported results, then added to the leaderboard by maintainers via data-file commits"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: MIT, Copyright (c) 2025 SHI-Labs."
    },
    "license_data": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "PAI-Bench-G: CC-BY-NC-4.0",
       "level": "verified",
       "sources": [
        "s9"
       ],
       "note": "HF dataset card metadata. Not gated."
      },
      {
       "value": "PAI-Bench-C: MIT; PAI-Bench-U: MIT",
       "level": "verified",
       "sources": [
        "s10"
       ],
       "note": "HF dataset card metadata for both (https://huggingface.co/datasets/shi-labs/physical-ai-bench-understanding). Source clips come from third-party datasets and the web; upstream terms are not restated on the cards."
      }
     ]
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "display": "non-commercial for G data; allowed for code, C and U by their stated licences",
     "note": "Reading of licences; not legal advice."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Only agreement with human raters was measured: Pearson r = 0.918 between PAI-Bench-G scores and Elo from a pairwise human study (source videos + 8 models), by the benchmark authors. No study links PAI-Bench scores to robot policy success. Looked in the paper, NVIDIA Cosmos-Predict2.5 and Cosmos 3 reports."
    },
    "citations": {
     "value": 34,
     "display": "34 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Batch API query for arXiv:2512.01989."
    },
    "github_stars": {
     "value": 103,
     "display": "103",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "used_by": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "NVIDIA Cosmos-Predict2.5 report reports PAI-Bench T2W/I2W results (calls it PAI-Bench-Predict).",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "NVIDIA Cosmos 3 report reports PAIBench-G T2V/I2V results; Cosmos-HumanEval is built on the PAIBench-G prompt set.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      }
     ]
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboard updated 2026-08-18; code fixes 2026-06."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Per-track counts 1,044 + 600 + 604 + 610 = 2,858, not the stated 2,808. unit of 'case' per",
     "text": "Our arithmetic; flagged, not resolved.",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "other",
     "title": "Two models score above the real-video reference row on the generation leaderboard (83.9 an",
     "text": "Reading of leaderboard data; suggests a ceiling on the automatic metric.",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "status": "open"
    },
    {
     "id": "i3",
     "type": "other",
     "title": "T2v generators span ~4 points on paibench-g vs ~10 on its human eval.",
     "text": "Claim by NVIDIA Cosmos 3 report (a supporter, not the PAI-Bench authors). Not answered by the PAI-Bench authors as far as we found.",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "PAI-Bench: A Comprehensive Benchmark For Physical AI (full text)",
     "url": "https://arxiv.org/html/2512.01989v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s2": {
     "title": "SHI-Labs/physical-ai-bench on GitHub (repository)",
     "url": "https://github.com/SHI-Labs/physical-ai-bench",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "PAI-Bench: A Comprehensive Benchmark For Physical AI",
     "url": "https://arxiv.org/abs/2512.01989",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s4": {
     "title": "shi-labs/physical-ai-bench-leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/shi-labs/physical-ai-bench-leaderboard/commits/main",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "CVPR 2026 Open Access Repository",
     "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Zhou_PAI-Bench_A_Comprehensive_Benchmark_For_Physical_AI_CVPR_2026_paper.html",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2606.02800v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    },
    "s7": {
     "title": "shi-labs/physical-ai-bench-leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/shi-labs/physical-ai-bench-leaderboard/tree/main/data",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s8": {
     "title": "SHI-Labs/physical-ai-bench on GitHub (blob)",
     "url": "https://github.com/SHI-Labs/physical-ai-bench/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "shi-labs/physical-ai-bench-generation on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/shi-labs/physical-ai-bench-generation",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s10": {
     "title": "shi-labs/physical-ai-bench-conditional-generation on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/shi-labs/physical-ai-bench-conditional-generation",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s11": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s12": {
     "title": "World Simulation with Video Foundation Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2511.00062v2",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s13": {
     "title": "shi-labs/physical-ai-bench-leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/shi-labs/physical-ai-bench-leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "partnr",
   "name": "PARTNR",
   "full_name": "PARTNR: A Benchmark for Planning and Reasoning in Embodied Multi-agent Tasks",
   "aliases": [
    "Planning And Reasoning Tasks in humaN-Robot collaboration",
    "partnr-planner"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "PARTNR is a Meta FAIR benchmark of 100,000 language-instructed household tasks in 60 simulated houses, where a planner controls a simulated Spot robot working with a simulated or real person. It mainly tests LLM task planning and coordination, scored by automatically generated checks.",
    "sources": [
     "s1",
     "s2"
    ],
    "short": "PARTNR is a set of 100,000 simulated household tasks in which a robot's planner (the model that decides its next steps) works with a person. It is mainly used to test task planning and coordination by large language models (LLMs)."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The paper and project page call PARTNR a benchmark."
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "Also a dataset: task episodes, human-in-the-loop traces, skill checkpoints and precomputed scene graphs.",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10"
    },
    "publishers": {
     "value": [
      "Meta FAIR"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "All 20 authors are marked 'Work done at FAIR Meta'; authors are listed alphabetically. The project page calls it 'A Meta FAIR Release'.",
     "short": "Meta FAIR"
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Meta FAIR is an industry AI research lab."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "Meta Platforms' principal executive offices are in Menlo Park, California (FY2025 10-K cover page). The paper does not say where the team sits."
    },
    "first_release": {
     "value": "2024-10",
     "display": "arXiv v1 and the Meta blog post on 2024-10-31. The episode dataset's 'initial release' commit is dated 2024-10-30.",
     "level": "verified",
     "sources": [
      "s1",
      "s5",
      "s9",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Code repository created 2024-10-28 (GitHub API). arXiv has only v1.",
     "short": "October 2024, at ICLR 2025"
    },
    "published_at": {
     "value": "ICLR 2025",
     "display": "ICLR 2025 (poster)",
     "level": "verified",
     "sources": [
      "s3",
      "s32"
     ],
     "checked": "2026-10-10",
     "short": "ICLR 2025"
    },
    "latest_update": {
     "value": "2025-07",
     "display": "Episode dataset: task-type metadata added 2025-07-16. Code: last commit on main 2025-04-17.",
     "level": "verified",
     "sources": [
      "s12",
      "s10",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub API's pushed_at of 2026-04-08 comes from automated dependency-update branches, not main.",
     "short": "July 2025. Task-type metadata was added to the dataset."
    },
    "version": {
     "value": "episodes v0_0",
     "display": "Episode dataset v0_0, the only entry in its changelog. Code has no tagged releases; pull request 'PARTNR version v0.1.0' was merged on 2025-01-31.",
     "level": "verified",
     "sources": [
      "s11",
      "s29",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "setup.py gives version '1.0'.",
     "short": "Episodes v0_0. The code has no tagged release."
    },
    "capability": {
     "value": [
      "collaboration",
      "long-horizon",
      "instruction-following",
      "embodied-reasoning",
      "mobile-manipulation"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Four task types: constraint-free, spatial, temporal and heterogeneous (one agent cannot do some actions). Tasks average 4.7 propositions. The object under test is mainly a high-level planner that calls fixed skills; the taxonomy cannot mark this."
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "new-task"
     ],
     "display": "Validation and test episodes use houses not in the training split, each with its own new instruction. Partners can be unseen real people.",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "100,000 train episodes in 37 scenes, 1,000 validation in 13, 1,000 test in 10 (paper Section 3.3).",
     "short": "New houses and new instructions"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Habitat 3.0 with HSSD scenes. Real people can control the human avatar in the same simulator."
    },
    "simulator": {
     "value": "Habitat 3.0",
     "display": "Habitat 3.0 (habitat-sim and habitat-lab), with HSSD scenes extended by articulated furniture.",
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "The paper says the code depends on v0.3.2; the installation guide installs habitat-sim 0.3.3. Meta stopped maintaining habitat-lab after v0.3.4 (2026-05-07).",
     "short": "Habitat 3.0"
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "display": "Simulated Spot robot with fixed skills (explore, navigate, pick, place, open, close). The partner is a simulated person driven by an LLM, or a real person.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Changed from the basic entry, which also listed 'humanoid'. The humanoid is a simulated person, not a robot under test."
    },
    "robots": {
     "value": "Boston Dynamics Spot (simulated)",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "Spot (simulated)"
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 100000,
     "display": "100,000 tasks (abstract). Paper splits: 100,000 train episodes, 1,000 validation, 1,000 test. Released: 111,652 train episodes verified by people (131,991 unverified) and 1,000 validation. No test split is released.",
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT between the paper's split sizes and the released files. The README lists only train_2k, val, train and val_mini as runnable splits.",
     "short": "100,000 tasks in the paper"
    },
    "scenes": {
     "value": 60,
     "display": "60 HSSD houses with added articulated furniture",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "60 houses"
    },
    "objects": {
     "value": 5819,
     "display": "5,819 unique objects from the OVMM object set. Tasks refer to 155 object types, 20 furniture classes and 13 room types.",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "5,819 objects"
    },
    "demonstrations": {
     "value": "Human-in-the-loop traces",
     "display": "De-identified traces from 129 participants solving tasks in the simulator (released for validation and a 2,000-episode train subset), plus ReAct traces used for retrieval and fine-tuning.",
     "level": "verified",
     "sources": [
      "s2",
      "s11",
      "s12"
     ],
     "checked": "2026-10-10",
     "short": "Traces from 129 people solving tasks in the simulator"
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Success means all propositions are satisfied under their constraints. Percent complete is the share of propositions satisfied."
    },
    "metric_detail": {
     "value": "Success and percent complete, from generated checks",
     "display": "Each episode has a Python evaluation function, generated by an LLM (CodeLlama-70B) from human-verified examples, that checks propositions, their order and constraints over the whole episode. It returns success, percent complete and a failure explanation. Papers also report simulation steps, planning cycles (capped at 50), task offloading, extraneous effort and exploration efficiency.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Manual check of 400 sampled generated episodes: 92% of evaluation functions and 90% of instructions correct, 83% both (Table 7). Validation and test episodes were annotated by people.",
     "short": "Success and share of the task completed, checked by code written by an LLM"
    },
    "trials": {
     "value": "1 run per episode on 1,000 validation episodes",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Main results",
       "display": "Table 2 and Table 10 give mean and standard error over the validation set (1,000 episodes). Table 11 gives the test set.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Human-in-the-loop",
       "display": "Each task was attempted up to 3 times, with a written explanation of the failure after each try. The successful try, or else the best try, was kept.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "1 run on each of 1,000 validation episodes"
    },
    "uncertainty_reported": {
     "value": "yes",
     "level": "inferred",
     "sources": [
      "s2",
      "s20",
      "s21",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "The paper reports mean and standard error, MIT Lincoln Lab one standard deviation and AHAT ±. FLEET's table gives single numbers."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s4",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s4",
      "s6",
      "s11",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard or challenge on the project page, README, dataset card or in the paper.",
     "short": "Results are reported only in papers."
    },
    "human_baseline": {
     "value": 0.93,
     "display": "People: success 0.93 alone or in pairs. A real person with an LLM-controlled robot: 0.91 to 0.92. LLM planners with no privileged information and an LLM partner: 0.30.",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Single person",
       "display": "Success 0.93±0.01, percent complete 0.96, 3046.99 steps",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Two people",
       "display": "Success 0.93±0.01, 2369.55 steps; the second person did 59% of sub-tasks",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Person with LLM robot",
       "display": "ReAct (Llama-3.1-70B): 0.91±0.01, 4267.71 steps, robot did 16%. Fine-tuned Llama-3.1-8B: 0.92±0.01, 3443.33 steps, robot did 26%. Both slower than a person alone.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "LLM planners, no privileged information",
       "display": "Learned skills plus ConceptGraphs perception, LLM partner: 0.30±0.01 in Table 2, 0.25±0.01 in Table 10 for the same setting.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "CONFLICT inside the paper; see issues.i2."
      },
      {
       "value": "Protocol differences",
       "display": "Human runs allowed up to 3 tries with failure feedback and kept the best. Tasks no person could solve in 6 tries were removed from the dataset. The human study used validation tasks; 129 non-expert US participants recruited by a third-party company.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "People succeed 0.93 of the time. LLM planners succeed 0.30 of the time without privileged information (information a real robot would not have)."
    },
    "top_score": {
     "value": "No single headline score",
     "display": "No single headline score. In the paper's default setting (two LLM agents, oracle skills, partial observability) ReAct with Llama-3.1-70B reaches success 0.73 on validation and 0.63 on test. Later papers use other settings and metric definitions.",
     "level": "verified",
     "sources": [
      "s2",
      "s20",
      "s21",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "Rows are not comparable: splits, skills, task subsets and metric definitions differ.",
     "items": [
      {
       "value": "Paper: heuristic expert (privileged)",
       "display": "Success 0.84±0.01 (validation), 0.69±0.02 (test)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Paper: ReAct, decentralized, oracle skills",
       "display": "Success 0.73±0.01 (validation), 0.63±0.02 (test)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Paper: fine-tuned Llama-3.1-8B",
       "display": "Success 0.70±0.01 (validation), 0.51±0.02 (test); 8.6 times faster than the 70B model",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Paper: learned skills",
       "display": "ReAct success 0.57±0.02 (validation), 0.50±0.02 (test)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "MIT Lincoln Lab, 2025-07",
       "display": "o3-mini 'success rate' 0.77 to 0.81 across four settings, with redefined metrics (issues.i6)",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "FLEET, 2025-10",
       "display": "Success 0.59 averaged over three task types, against 0.28 for the PARTNR decentralized baseline (GPT-4o and gpt-oss-20b runs)",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "AHAT / TGPO, 2026",
       "display": "High-level planning success 82.6% (GPT-5: 69.2%), planning only",
       "level": "verified",
       "sources": [
        "s22"
       ]
      }
     ],
     "short": "There is no single headline score. ReAct with Llama-3.1-70B reaches 0.73 in the paper's default setting."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s7",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, 'Copyright (c) Meta Platforms, Inc. and its affiliates'.",
     "short": "MIT"
    },
    "license_data": {
     "value": [
      "CC-BY-NC-4.0"
     ],
     "display": "Episodes, checkpoints, scene graphs and human traces: CC BY-NC 4.0 (dataset card).",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "short": "CC BY-NC 4.0"
    },
    "license_assets": {
     "value": "CC-BY-NC-4.0 and unlabelled",
     "display": "HSSD scenes (partnr branch) CC BY-NC 4.0. OVMM objects have no card or licence file. Humanoid avatars non-commercial, with conflicting labels. Spot model by permission of Boston Dynamics.",
     "level": "verified",
     "sources": [
      "s14",
      "s15",
      "s16",
      "s17",
      "s8"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "OVMM objects",
       "level": "unknown",
       "sources": [],
       "note": "The Hugging Face repo has no card or licence file. Folder names show objects from AI2-THOR, Amazon Berkeley Objects, Google Scanned Objects and HSSD; we did not check those sources' licences."
      },
      {
       "value": "Humanoid avatars",
       "display": "Card metadata says CC BY-NC-SA 4.0; card text says CC BY-NC 4.0, with walking motion under the SMPL Body Motion File License.",
       "level": "verified",
       "sources": [
        "s16"
       ]
      }
     ],
     "short": "HSSD scenes are non-commercial. The objects have no licence label."
    },
    "access": {
     "value": "open",
     "display": "Code on GitHub; data on Hugging Face without gating.",
     "level": "verified",
     "sources": [
      "s13",
      "s8",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Running the paper's baselines needs Llama-3.1-70B; the human study hosted each 70B model on 4 A100 GPUs. The HSSD card shows a licence acknowledgement prompt.",
     "short": "Open download"
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s11",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Episodes and scenes are CC BY-NC 4.0. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No real-robot comparison found.",
     "level": "inferred",
     "sources": [
      "s2",
      "s5",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "The paper has no real-robot runs. Meta's blog calls testing in physical-world scenarios a future goal. FLEET ran two real Spot robots on different inspection tasks, with no paired PARTNR comparison.",
     "short": "Not checked on real robots"
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "citations": {
     "value": 98,
     "display": "98 (Semantic Scholar; 15 influential)",
     "level": "verified",
     "sources": [
      "s18"
     ],
     "checked": "2026-10-10",
     "short": "98"
    },
    "github_stars": {
     "value": 394,
     "display": "394 stars, 53 forks (facebookresearch/partnr-planner)",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "short": "394"
    },
    "dataset_downloads": {
     "value": 609,
     "display": "609 (Hub 'downloads' field), 12,129 all time, 7 likes",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "We did not check the time window behind the 'downloads' field.",
     "short": "609 on Hugging Face"
    },
    "used_by": {
     "value": "A few third-party papers report PARTNR results, each with its own protocol.",
     "level": "verified",
     "sources": [
      "s20",
      "s21",
      "s22",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar lists 98 citing papers; we opened those whose citation contexts suggested results on PARTNR.",
     "items": [
      {
       "value": "MIT Lincoln Laboratory",
       "display": "2025-07: GPT-4o, o3-mini, Llama 3 and DeepSeek-Llama-70B as planners.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "JHU APL and DEVCOM ARL (FLEET)",
       "display": "2025-10: scheduling for two-robot teams on three task types.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Shanghai Innovation Institute et al. (AHAT / TGPO)",
       "display": "2026-02, v2 2026-07: planning-only evaluation.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      }
     ],
     "short": "A few third-party papers"
    },
    "derived_benchmarks": {
     "value": [
      "PARTNR-Dialog",
      "PARTNR with dialogue (2605.12920)"
     ],
     "level": "verified",
     "sources": [
      "s23",
      "s24"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "PARTNR-Dialog",
       "display": "2026-06. Communicative cooperative planning benchmark built on the PARTNR environment (LLawCo paper).",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "PARTNR with a dialogue channel",
       "display": "2026-05. Two partially observing LLM agents exchange messages during PARTNR tasks.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      }
     ],
     "short": "Two variants with dialogue"
    },
    "status": {
     "value": "dormant",
     "display": "No code changes on main since 2025-04-17; dataset metadata last changed 2025-07-16. Issues opened since 2025-09 have no maintainer reply.",
     "level": "inferred",
     "sources": [
      "s10",
      "s12",
      "s26",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "It also depends on habitat-lab, which Meta stopped maintaining in 2026-05.",
     "short": "No updates since mid-2025"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "People and LLMs are scored under different rules",
     "text": "The headline 'humans solve 93%, LLMs 30%' compares different set-ups. In the human runs each task could be tried up to 3 times, with a written explanation of what went wrong after each try, and the best try was kept. Tasks that no person solved in 6 tries had already been removed from the dataset. The 0.30 comes from LLM planners with learned skills, scene graphs built from camera images and an LLM-controlled partner, in one run. When the partner is a real person, the same kind of LLM robot team succeeds 0.91 to 0.92 of the time.",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "status": "open",
     "short": "People could retry each task and got feedback after each try. LLM agents got one run."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The headline LLM number differs between tables",
     "text": "For ReAct with learned skills and ConceptGraphs perception on the validation set, Table 2 gives success 0.30±0.01 and Table 10 gives 0.25±0.01, with different step counts (12490.80 against 12274.27). The project page quotes 30%. Test-set results (Table 11) are lower than validation throughout: for example 0.51 against 0.70 for the fine-tuned model and 0.69 against 0.84 for the heuristic expert.",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "status": "open",
     "short": "The paper reports the same setting as 0.30 in one table and 0.25 in another. Scores on the test set are lower than on the validation set."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Released data does not match the paper",
     "text": "The paper describes 100,000 train, 1,000 validation and 1,000 test episodes. The release has 111,652 people-verified train episodes (131,991 unverified) and 1,000 validation episodes, and no test split, so the paper's Table 11 cannot be reproduced. The paper says weights of the fine-tuned planner were released; the dataset holds only the skill checkpoints and an issue reporting this has had no reply since 2025-11-22. Another open issue reports that the code for task offloading and related metrics covers only the human runs. The constrained decoding library does not support Qwen or Llama-3.2 and newer models.",
     "level": "verified",
     "sources": [
      "s2",
      "s11",
      "s12",
      "s26",
      "s27",
      "s28"
     ],
     "status": "open",
     "short": "The release has no test split. The weights of the fine-tuned planner are missing."
    },
    {
     "id": "i4",
     "type": "shortcut",
     "title": "Some tasks are already done at the start",
     "text": "Users report validation episode 139, where the objects already met the goal and an agent succeeded by checking and calling Done. The paper's own error analysis lists 'Already Satisfied' as a task-generation failure (2% of sampled constraint-free, spatial and temporal episodes).",
     "level": "verified",
     "sources": [
      "s25",
      "s2"
     ],
     "status": "contested",
     "counter": {
      "text": "A PARTNR co-author replied that such episodes are intentional, to test whether agents can verify that a task is complete, and that filtering keeps their number limited. An impossible test episode was fixed.",
      "sources": [
       "s25"
      ],
      "short": "A co-author says these episodes are intentional and that there are few of them."
     },
     "short": "Some episodes are already solved at the start. The authors say this is intended."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Generated checks are sometimes wrong",
     "text": "Evaluation functions are written by an LLM. On 400 sampled generated episodes, 92% of evaluation functions and 83% of task and check pairs were correct; for spatial tasks only 74% were. Errors include wrong ordering constraints and wrong predicates. Validation and test episodes were annotated by people, but their residual error rate is not reported.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "About 1 in 6 generated training episodes has a mistake in the task or in its automatic check."
    },
    {
     "id": "i6",
     "type": "inconsistent-reporting",
     "title": "Later papers change the metrics or the task set",
     "text": "MIT Lincoln Lab redefines success rate as the share of successful decision points and percent complete as the share of successful episodes, ran each setting for ten hours (different episode counts per model), and reprints the paper's Llama-3.1-70B rows next to its own. FLEET reports three of the four task types. AHAT scores planning only. The README notes that constrained generation, which the paper uses to block invalid actions, is not supported for OpenAI models.",
     "level": "verified",
     "sources": [
      "s20",
      "s21",
      "s22",
      "s6"
     ],
     "status": "open",
     "short": "Papers by other groups redefine the metrics or use only some of the tasks."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "PARTNR measures high-level planning and coordination with fixed skills in simulation. A high score says little about low-level manipulation or real robots. With learned skills and realistic perception, the best LLM success fell to 0.30 (or 0.25) in the paper.",
     "basis": [
      "facts.human_baseline",
      "facts.sim_to_real",
      "facts.top_score"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "PARTNR tests planners in simulation. It does not test robot control."
    },
    {
     "id": "r2",
     "text": "The gap between people and LLM agents is real, but the headline numbers overstate it. People had retries and feedback, and the automated runs pair the LLM robot with a weaker LLM partner.",
     "basis": [
      "issues.i1",
      "facts.human_baseline"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "People do better than LLM agents, but part of the gap comes from differences in how they were tested."
    },
    {
     "id": "r3",
     "text": "Do not compare PARTNR numbers across papers without checking the split, the skills, the perception setting and the metric definitions. The released data supports only validation-set comparisons.",
     "basis": [
      "issues.i2",
      "issues.i3",
      "issues.i6",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the split, the skills and the metric definitions before you compare numbers across papers."
    }
   ],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "PARTNR paper full text, project page, Meta blog (2024-10-31), dataset card, README, Semantic Scholar citation contexts (98 records) filtered for real-robot mentions, FLEET (only paper with hardware trials), web searches. None found.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project page, README, dataset card, ICLR page, paper text.",
     "date": "2026-10-10"
    },
    {
     "for": "used_by",
     "where": "Semantic Scholar citation contexts (98 records); opened 2507.06157, 2510.07417, 2602.12244, 2605.12920, 2606.28182. OpenAlex could not be queried (daily budget exhausted); arXiv API search returned errors.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (OVMM objects)",
     "where": "ai-habitat/OVMM_objects repository tree: no card, no licence file.",
     "date": "2026-10-10"
    },
    {
     "for": "issues (reviews)",
     "where": "OpenReview forum T5QLRRHyL1; the API returned HTTP 403, so reviews were not read.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "PARTNR: A Benchmark for Planning and Reasoning in Embodied Multi-agent Tasks (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2411.00081",
     "type": "paper",
     "publisher": "arXiv (Meta FAIR)",
     "date": "2024-10-31",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "PARTNR paper, full text v1 (Sections 3 and 4; Appendices A.1, A.6, A.11, A.13)",
     "url": "https://arxiv.org/html/2411.00081v1",
     "type": "paper",
     "publisher": "arXiv (Meta FAIR)",
     "date": "2024-10-31",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "PARTNR, ICLR 2025 proceedings page",
     "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/a3cf318fbeec1126da21e9185ae9908c-Abstract-Conference.html",
     "type": "paper",
     "publisher": "ICLR 2025",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "PARTNR project page",
     "url": "https://aihabitat.org/partnr/",
     "type": "site",
     "publisher": "Meta FAIR (AI Habitat)",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Advancing embodied AI through progress in touch perception, dexterity, and human-robot interaction (PARTNR section)",
     "url": "https://ai.meta.com/blog/fair-robotics-open-source/",
     "type": "blog",
     "publisher": "Meta AI",
     "date": "2024-10-31",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "facebookresearch/partnr-planner README",
     "url": "https://github.com/facebookresearch/partnr-planner/blob/main/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "partnr-planner LICENSE",
     "url": "https://github.com/facebookresearch/partnr-planner/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "partnr-planner INSTALLATION.md (habitat-sim 0.3.3, data downloads)",
     "url": "https://github.com/facebookresearch/partnr-planner/blob/main/INSTALLATION.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "GitHub API: facebookresearch/partnr-planner (stars, forks, created, pushed) and commit list",
     "url": "https://api.github.com/repos/facebookresearch/partnr-planner",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "partnr-planner commit history (main)",
     "url": "https://github.com/facebookresearch/partnr-planner/commits/main",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2025-04-17",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "ai-habitat/partnr_episodes dataset card (splits, licence, changelog)",
     "url": "https://huggingface.co/datasets/ai-habitat/partnr_episodes",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2025-07-16",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Hugging Face Hub API: ai-habitat/partnr_episodes file tree and commits",
     "url": "https://huggingface.co/api/datasets/ai-habitat/partnr_episodes/tree/main?recursive=true",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Hugging Face Hub API record for ai-habitat/partnr_episodes (downloads, gated flag)",
     "url": "https://huggingface.co/api/datasets/ai-habitat/partnr_episodes?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "hssd/hssd-hab dataset card (CC BY-NC 4.0)",
     "url": "https://huggingface.co/datasets/hssd/hssd-hab",
     "type": "dataset",
     "publisher": "HSSD authors on Hugging Face",
     "date": "2025-02-14",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Hugging Face Hub API: ai-habitat/OVMM_objects file tree (no card, no licence file)",
     "url": "https://huggingface.co/api/datasets/ai-habitat/OVMM_objects/tree/main/train_val",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "ai-habitat/habitat_humanoids dataset card",
     "url": "https://huggingface.co/datasets/ai-habitat/habitat_humanoids",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2023-10-18",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "ai-habitat/hab_spot_arm dataset card (Spot URDF licence note)",
     "url": "https://huggingface.co/datasets/ai-habitat/hab_spot_arm",
     "type": "dataset",
     "publisher": "AI Habitat (Meta) on Hugging Face",
     "date": "2025-02-14",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Semantic Scholar API record for arXiv:2411.00081",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2411.00081?fields=title,citationCount,influentialCitationCount,externalIds,venue,year,publicationDate",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Semantic Scholar API: citations of arXiv:2411.00081 with contexts (98 records)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2411.00081/citations?fields=title,externalIds,year,publicationDate,contexts,intents&limit=100",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Evaluation of Habitat Robotics using Large Language Models (Table II)",
     "url": "https://arxiv.org/html/2507.06157",
     "type": "paper",
     "publisher": "arXiv (MIT Lincoln Laboratory)",
     "date": "2025-07-08",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "FLEET: Formal Language-Grounded Scheduling for Heterogeneous Robot Teams (Table I, Section IV-D)",
     "url": "https://arxiv.org/html/2510.07417",
     "type": "paper",
     "publisher": "arXiv (JHU APL, DEVCOM ARL)",
     "date": "2025-10-08",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Any House Any Task (v2 retitled TGPO), Table 1",
     "url": "https://arxiv.org/html/2602.12244",
     "type": "paper",
     "publisher": "arXiv (Shanghai Innovation Institute, Shanghai Jiao Tong University et al.)",
     "date": "2026-02-12",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior (PARTNR-Dialog)",
     "url": "https://arxiv.org/abs/2606.28182",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06-26",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Embodied Multi-Agent Coordination by Aligning World Models Through Dialogue",
     "url": "https://arxiv.org/abs/2605.12920",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05-13",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "partnr-planner issue #18: episodes already satisfied at the start (with co-author reply)",
     "url": "https://github.com/facebookresearch/partnr-planner/issues/18",
     "type": "repo",
     "publisher": "facebookresearch/partnr-planner",
     "date": "2025-03-10",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "partnr-planner issue #38: fine-tuned planning model weights missing",
     "url": "https://github.com/facebookresearch/partnr-planner/issues/38",
     "type": "repo",
     "publisher": "facebookresearch/partnr-planner (user report)",
     "date": "2025-11-22",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "partnr-planner issue #35: evaluation metrics code missing for LLM-LLM runs",
     "url": "https://github.com/facebookresearch/partnr-planner/issues/35",
     "type": "repo",
     "publisher": "facebookresearch/partnr-planner (user report)",
     "date": "2025-09-25",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "partnr-planner issue #43: transformers-CFG does not support Qwen or Llama-3.2 and above",
     "url": "https://github.com/facebookresearch/partnr-planner/issues/43",
     "type": "repo",
     "publisher": "facebookresearch/partnr-planner (user report)",
     "date": "2026-03-18",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "partnr-planner pull request #6: PARTNR version v0.1.0 (merged)",
     "url": "https://github.com/facebookresearch/partnr-planner/pull/6",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2025-01-31",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "habitat-lab README (Meta maintenance notice beyond v0.3.4)",
     "url": "https://github.com/facebookresearch/habitat-lab/blob/main/README.md",
     "type": "repo",
     "publisher": "Meta (facebookresearch)",
     "date": "2026-05-07",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Meta Platforms, Inc. Form 10-K for fiscal year 2025 (cover page: principal executive offices)",
     "url": "https://www.sec.gov/Archives/edgar/data/1326801/000162828026025534/meta-12312025x10kars.htm",
     "type": "report",
     "publisher": "Meta Platforms, Inc. (SEC filing)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "ICLR 2025 poster page: PARTNR",
     "url": "https://iclr.cc/virtual/2025/poster/29562",
     "type": "paper",
     "publisher": "ICLR 2025",
     "date": "2025-04",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and research/raw/inventory/frontier-labs.json. Changes from the basic entry: embodiment no longer lists 'humanoid'; prior claim 'humans 0.93 vs LLM 0.30' confirmed with caveats (retries for people, 0.25 in Table 10, different partners); added test-set results, missing planner weights, the 'already satisfied' dispute, third-party results, issues and readings. Checks ran on 2026-10-10 and into early 2026-10-11 local time."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well the planner will work on a real robot.",
     "sub": "We found no runs on real robots.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How good the robot is at low-level manipulation (moving and handling objects).",
     "sub": "The robot's skills are either oracle skills provided by the simulator or fixed learned modules.",
     "basis": [
      "facts.embodiment",
      "facts.top_score"
     ]
    },
    {
     "id": "l3",
     "text": "How a planner does on the test set.",
     "sub": "The test episodes have not been released.",
     "basis": [
      "issues.i3"
     ]
    }
   ],
   "validity": []
  },
  {
   "id": "polaris",
   "name": "PolaRiS",
   "full_name": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies",
   "aliases": [
    "Policy Evaluation and Environment Reconstruction in Simulation",
    "PolaRiS Hub",
    "polaris-evals"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Simulated evaluation environments that score real-data-trained generalist manipulation policies, validated against paired real-robot runs; a real-to-sim benchmark plus a toolkit for building more environments.",
   "summary": {
    "text": "PolaRiS turns short video scans of real scenes into simulated test environments (Gaussian splats with Isaac Sim physics) for scoring generalist robot policies on the DROID Franka setup. Its authors report a mean Pearson correlation of 0.90 with real-robot results over 5 policies and 6 scenes (0.83 in the RSS 2026 version), after each policy is fine-tuned briefly on simulated data.",
    "short": "PolaRiS builds simulated test scenes from short video scans of real scenes, to score robot policies (the robots' control models) for the DROID robot setup. Its authors report that the scores track real-robot results after each policy is briefly fine-tuned with some simulated data (co-training).",
    "sources": [
     "s2",
     "s3"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Ships 6 fixed evaluation environments with initial conditions and scoring rubrics, plus tools to build more. Classification by the Atlas."
    },
    "kind_secondary": {
     "value": [
      "platform"
     ],
     "display": "Also a toolkit (video scan, Gaussian-splat reconstruction, web scene composer) and a hub for sharing new environments",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The taxonomy has no value for an environment-building toolkit; 'platform' is the closest fit."
    },
    "version": {
     "value": "0.1.0",
     "display": "Package polaris 0.1.0. No tags or releases. Paper: arXiv v1 (2025-12-18) and v2 (2025-12-30, references and acknowledgements only), then the RSS 2026 version with revised numbers.",
     "level": "verified",
     "sources": [
      "s9",
      "s10",
      "s1",
      "s3"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "arXiv v2 vs RSS 2026",
       "display": "arXiv: mean r 0.90, MMRV 0.03, 'two institutions'. RSS: r 0.83 in the text (0.84 in Fig. 6), MMRV 0.08, bootstrap intervals, a new physics-randomisation test, 600 real and over 93,000 simulated rollouts.",
       "level": "verified",
       "sources": [
        "s2",
        "s3"
       ]
      },
      {
       "value": "Hub data change 2026-03-14",
       "display": "New initial conditions for the three Princeton environments (commit 'fix princeton ics'), no changelog. Copies downloaded earlier hold the old files.",
       "level": "verified",
       "sources": [
        "s13",
        "s15"
       ]
      }
     ],
     "short": "0.1.0. There are no tagged releases."
    },
    "publishers": {
     "value": [
      "University of Washington",
      "Princeton University",
      "UC Berkeley",
      "Stanford University",
      "Toyota Research Institute",
      "University of Southern California",
      "Cornell University",
      "Physical Intelligence"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "14 authors. Equal contribution: Arhan Jain (UW), Mingtong Zhang (Princeton). Equal advising: Abhishek Gupta (UW, TRI), Karl Pertsch (Berkeley, Stanford, Physical Intelligence). Sergey Levine and Chelsea Finn also list Physical Intelligence."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Six universities plus two companies (TRI, Physical Intelligence) through author affiliations. All five tested policies come from Physical Intelligence's openpi DROID line."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All affiliations are in the United States."
    },
    "first_release": {
     "value": "2025-12",
     "display": "arXiv v1 on 2025-12-18. Published at RSS 2026 (Robotics: Science and Systems XXII, Sydney, 13-17 July 2026).",
     "level": "verified",
     "sources": [
      "s1",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Code repository created 2025-11-24; Hub dataset created 2025-11-29.",
     "short": "December 2025, at RSS 2026"
    },
    "latest_update": {
     "value": "2026-07",
     "display": "2026-07-13: README fix merged in the code repo; July 2026: RSS version with revised correlation numbers. Last data change: 2026-03-14 (Princeton initial conditions).",
     "level": "verified",
     "sources": [
      "s10",
      "s3",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Code commits since January 2026 are typo and documentation fixes.",
     "short": "July 2026. A fix to the README was merged."
    },
    "status": {
     "value": "maintained",
     "display": "Fixes only since January 2026. On 2026-06-24 the lead author wrote that a larger release with more environments is in preparation; the Princeton reset states are still broken.",
     "level": "inferred",
     "sources": [
      "s10",
      "s16",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "9 open issues on 2026-10-10.",
     "short": "Only fixes since January 2026. A larger release has been announced."
    },
    "capability": {
     "value": [
      "manipulation"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Rigid-object pick-and-place, stacking and wiping tasks with one language instruction per scene."
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "object-pose"
     ],
     "display": "The 6 evaluation scenes are unseen by the sim co-training step; object start positions vary within each scene.",
     "level": "inferred",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Co-training scenes and evaluation scenes share no scenes or objects (paper Section 5.1). Whether the DROID pretraining data contain similar scenes is not stated.",
     "short": "Unseen scenes and varied start positions"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Real-to-sim: reconstructions of real scenes."
    },
    "simulator": {
     "value": "Isaac Sim via Isaac Lab 2.3.0, with 2D Gaussian splat rendering",
     "level": "verified",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "2DGS meshes give collision geometry and splats give images; objects are generated from photos with TRELLIS; scenes are composed in a web GUI and exported as USD. Policies use joint-position actions; DROID control runs at 15 Hz (Appendix D).",
     "short": "Isaac Sim, with Gaussian splat rendering (scenes rebuilt from video)"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "DROID platform: Franka Panda 7-DoF arm, Robotiq 2F-85 gripper, one wrist and one external ZED camera",
     "level": "verified",
     "sources": [
      "s2",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Gripper model read from the Hub's robot folder file names. Supports wrist cameras, unlike SimplerEnv's green-screening."
    },
    "scene": {
     "value": [
      "tabletop",
      "kitchen",
      "office-lab"
     ],
     "level": "inferred",
     "sources": [
      "s2",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Tabletop tasks on a kitchen table, a stovetop, a corner table and lab benches (from scene asset names). 3 scenes at UW and 3 at Princeton."
    },
    "tasks": {
     "value": 6,
     "display": "6 evaluation tasks, one per scene: block stacking, food bussing, pan cleaning, move latte cup, organize tools, tape into container. 15 further scenes are used only for co-training.",
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "6 tasks in 6 scenes"
    },
    "objects": {
     "value": 31,
     "display": "31 object assets across the 6 Hub environments (our count; scene backgrounds excluded)",
     "level": "inferred",
     "sources": [
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Counted from the Hub's per-environment asset folders on 2026-10-10."
    },
    "demonstrations": {
     "value": 316,
     "display": "316 simulated teleoperated episodes in the released co-training dataset (RLDS, 3.1 GB, 15 scenes). The paper's table lists 326 trajectories (56,041 timesteps) for this data mix; its text says about 350.",
     "level": "inferred",
     "sources": [
      "s14",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "CONFLICT. 316 is our sum of the shard lengths in dataset_info.json. Collected with a VR controller through the DROID teleoperation code. Each policy is co-finetuned on 10% of this data and 90% real DROID data for 1k steps.",
     "short": "316 simulated demonstrations for co-training"
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Each task has a step rubric (reach, lift, place) normalised to 0-1. Simulation scores from object states; real rollouts were scored by humans with the same rubric. The code also logs binary success."
    },
    "metric_detail": {
     "value": "normalised task progress, correlated with real-world progress",
     "display": "Main score: normalised task progress (0-1) averaged over rollouts after co-training. Agreement with real robots is reported as Pearson r and MMRV.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "An outside user asked whether the site's plots show progress or success (issue #8); the site states progress.",
     "short": "Task progress score from 0 to 1"
    },
    "trials": {
     "value": "50 simulated rollouts per task (paper)",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s16",
      "s21"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper",
       "display": "50 simulated rollouts per task; 20 real rollouts per policy and scene for validation.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "RSS version",
       "display": "Bootstrap: 50 episodes resampled per policy, 500 repeats, 95% intervals.",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "Public code",
       "display": "An outside user's run of the eval script covered 100 episodes per environment; the author says the UW scenes got 50 extra reset states for the release.",
       "level": "verified",
       "sources": [
        "s16"
       ]
      },
      {
       "value": "CoVer-VLA",
       "display": "50 episodes x 3 seeds on three PolaRiS environments.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      }
     ],
     "short": "50 per task in the paper"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s3",
      "s2",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "The RSS version adds bootstrap intervals for the correlation metrics; ablations show error bars over 5 seeds. Per-policy simulation scores are not given with intervals. CoVer-VLA reports standard deviations."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Users run the environments themselves."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The project site shows the paper's results; no submission page or ranking."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file, copyright 2025 Arhan Jain."
    },
    "license_data": {
     "value": "MIT",
     "display": "MIT for the PolaRiS Hub environments and the co-training dataset (Hugging Face cards)",
     "level": "verified",
     "sources": [
      "s12",
      "s14"
     ],
     "checked": "2026-10-10"
    },
    "license_assets": {
     "value": "MIT",
     "display": "MIT per the Hub card. Running PolaRiS needs Isaac Sim, whose Omniverse Kit SDK and NVIDIA 3D models are under NVIDIA's separate licence.",
     "level": "verified",
     "sources": [
      "s12",
      "s25",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "The Hub's robot folder (nvidia_droid: Franka arm and Robotiq gripper meshes) states no origin or separate terms; the folder name suggests NVIDIA (inferred)."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s7",
      "s12",
      "s14",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub; environments (<2 GB) and co-training data on ungated Hugging Face; co-trained checkpoints in Physical Intelligence's public Google Cloud bucket. Needs an NVIDIA GPU (tested on RTX 3090 and 5090) and Isaac Sim.",
     "short": "Open, on GitHub and Hugging Face"
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s8",
      "s12",
      "s25",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Code and data are MIT. Isaac Sim's additional components are under NVIDIA's licence, which limits use to Isaac Sim and Isaac Lab, forbids redistribution and modification, is revocable and does not address commercial use explicitly (read 2026-10-10). The robot model's origin is unstated. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Measured by its authors: mean Pearson r 0.90 over 6 scenes (arXiv), 0.83 with bootstrapping (RSS 2026), after co-training each policy on sim data. Without co-training the authors measured r 0.30; an independent group, also without co-training, got per-task r from -0.396 to 0.822 in its own scenes.",
     "level": "inferred",
     "sources": [
      "s2",
      "s3",
      "s20",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; each study is under validity. Not marked 'replicated': the independent study (SimFoundry) skipped PolaRiS's required co-training step and used its own scenes. The simulated policy is a co-finetuned version; real scores come from the original policies.",
     "short": "Measured by its authors. Each policy must first be fine-tuned on simulated data."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation. The validation itself used real scenes at two institutions."
    },
    "citations": {
     "value": 39,
     "display": "39 (Semantic Scholar; 4 influential)",
     "level": "verified",
     "sources": [
      "s19"
     ],
     "checked": "2026-10-10",
     "short": "39"
    },
    "github_stars": {
     "value": 238,
     "display": "238 stars, 34 forks (arhanjain/polaris)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "238"
    },
    "dataset_downloads": {
     "value": 802,
     "display": "802 (Hub 'downloads' field), 8,493 all time, 4 likes: PolaRiS-Hub",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hub API; time window of the 'downloads' field not checked. Co-training dataset: 754 and 3,632.",
     "short": "802 on Hugging Face"
    },
    "used_by": {
     "value": "CoVer-VLA (2026-02) reports results on three PolaRiS environments; SimFoundry (2026-06) uses it as a baseline; H2RBench (2026-09) is built on it; RLinf supports PolaRiS evaluation.",
     "level": "verified",
     "sources": [
      "s21",
      "s20",
      "s22",
      "s23"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "CoVer-VLA",
       "display": "Stanford and NVIDIA, 2026-02: pi0.5 with and without its verifier on three PolaRiS environments.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "SimFoundry",
       "display": "NVIDIA and others, 2026-06: baseline for real-to-sim evaluation.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "H2RBench",
       "display": "CMU, Michigan, TRI, 2026-09: environments built with PolaRiS on Isaac Lab.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      },
      {
       "value": "RLinf",
       "display": "Open RL framework with PolaRiS evaluation configs for openpi policies.",
       "level": "verified",
       "sources": [
        "s23"
       ]
      }
     ],
     "short": "A few papers and one reinforcement learning framework"
    },
    "industry_use": {
     "value": [
      "Physical Intelligence",
      "NVIDIA"
     ],
     "level": "verified",
     "sources": [
      "s24",
      "s21",
      "s20"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Physical Intelligence",
       "display": "openpi carries the PolaRiS training configs and PI's public bucket hosts the PolaRiS checkpoints. PI is also an author affiliation.",
       "level": "verified",
       "sources": [
        "s24",
        "s11"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "NVIDIA researchers used PolaRiS in CoVer-VLA and compared against it in SimFoundry.",
       "level": "verified",
       "sources": [
        "s21",
        "s20"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "H2RBench"
     ],
     "level": "verified",
     "sources": [
      "s22"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "H2RBench",
       "display": "2026-09. Human-to-robot transfer benchmark built with PolaRiS on Isaac Lab, with Marble replacing the scene reconstruction step.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      }
     ],
     "short": "1 benchmark built on PolaRiS"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Scores depend on fine-tuning each policy in simulation",
     "text": "PolaRiS's protocol co-finetunes every policy for 1k steps on 90% real DROID data and 10% simulated data before scoring it. The policy scored in simulation is therefore not the one deployed on the real robot. In the authors' own ablation, the same policies evaluated without co-training gave Pearson r 0.30 and MMRV 0.29 (co-trained: 0.86 and 0.07). Fine-tuning for too long also lowers correlation. Policies whose weights cannot be fine-tuned cannot be evaluated this way (our inference).",
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "status": "open",
     "short": "Each policy must first be fine-tuned on simulated data. Without this step, the correlation with real results was r = 0.30."
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "Three of the six public environments do not reproduce the paper",
     "text": "In March 2026 a user reported start states that made Move Latte Cup and Tape Into Container impossible. The author uploaded new initial conditions on 2026-03-14 without testing them; the user still saw objects stuck or falling off the table. In June 2026 another user ran the released pi0.5 checkpoint and got much lower scores than reported on the Princeton scenes: task progress 33.3% vs 60.0% reported on Organize Tools (0/100 successes), 34.3% vs 80.0% on Tape Into Container. On 2026-06-24 the author confirmed the Princeton reset states are broken, said the UW scenes should be fine, and said a fix will come with a larger release. The user also hit occasional crashes in the public code, which the author has seen too.",
     "level": "verified",
     "sources": [
      "s15",
      "s16",
      "s13"
     ],
     "status": "open",
     "short": "The author confirmed broken reset states in 3 of the 6 public scenes."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "Headline numbers differ between versions",
     "text": "arXiv v2 reports mean Pearson r 0.90 and MMRV 0.03 (the plain mean over six scenes). The RSS 2026 version reports r 0.83 in the text and 0.84 in Fig. 6, with MMRV 0.08, after bootstrap resampling. Comparison numbers changed too: Ctrl-World MMRV 0.22 to 0.23, LIBERO r 0.66 / 0.70 / 0.66 to 0.63 / 0.68 / 0.64. The released co-training data has 316 episodes, against 326 in the paper's table and about 350 in its text.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s14"
     ],
     "status": "open",
     "short": "The arXiv version reports a correlation of r = 0.90. The RSS version reports 0.83."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The validation set is small and narrow",
     "text": "The real-robot check used 5 policies from one family (openpi DROID policies, with pi0 at two training stages), 6 rigid-object tasks at two institutions, 20 real rollouts per policy and scene, and human-scored progress. Each per-scene correlation rests on 5 points. The authors say the tasks leave out non-rigid objects, soft-body interaction and complex contact, and that PolaRiS does not replace real-world evaluation.",
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "status": "open",
     "short": "The check against real robots used 5 related policies and 6 rigid-object tasks."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "An independent test found low correlation without co-training",
     "text": "SimFoundry's authors rebuilt their own DROID scenes with PolaRiS's tools and scored 5 policies (pi0, pi0.5, GR00T N1.6, GR00T N1.7, DreamZero) zero-shot, 25 rollouts per policy and task, real and simulated. Per-task Pearson r ranged from -0.396 to 0.822 (undefined on one task), MMRV from 0.053 to 0.352, and most policies scored far below their real success.",
     "level": "verified",
     "sources": [
      "s20"
     ],
     "status": "contested",
     "counter": {
      "text": "PolaRiS's protocol requires co-training, which SimFoundry skipped for both systems; the PolaRiS authors' own zero-shot ablation also gave low correlation (r 0.30). PolaRiS's co-trained pi0.5 scored higher in SimFoundry's scenes but was left out of the correlation.",
      "sources": [
       "s20",
       "s2"
      ],
      "short": "The outside test skipped the co-training step that PolaRiS requires."
     },
     "short": "An outside group skipped co-training and got correlations from -0.396 to 0.822, depending on the task."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "Treat PolaRiS scores as estimates of how DROID-style vision-language-action (VLA) models rank on rigid tabletop tasks. These estimates are valid only under the co-training protocol. The supporting evidence is the authors' study of 5 closely related policies. The one independent test skipped co-training and found low correlation, which matches the authors' own result without co-training (zero-shot).",
     "basis": [
      "facts.sim_to_real",
      "issues.i1",
      "issues.i4",
      "issues.i5"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "PolaRiS is useful for ranking DROID policies, but only with the co-training step."
    },
    {
     "id": "r2",
     "text": "Use the three UW scenes until the Princeton reset states are fixed. The author has confirmed that the Princeton reset states are broken, and an outside run scored far below the paper on two of those scenes.",
     "basis": [
      "issues.i2"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Use the UW scenes until the Princeton reset states are fixed."
    },
    {
     "id": "r3",
     "text": "PolaRiS's main value is that new test scenes are cheap to build (under 20 minutes of human time, under an hour in total), so teams can test in their own deployment scenes. Whether the correlation holds in a new scene has to be checked there.",
     "basis": [
      "facts.kind_secondary",
      "facts.sim_to_real"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "New scenes are cheap to build. The correlation must be checked again in each new scene."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How a policy performs without changes.",
     "sub": "Each policy is scored after 1k steps of fine-tuning on simulated data.",
     "basis": [
      "issues.i1",
      "facts.metric_detail"
     ]
    },
    {
     "id": "l2",
     "text": "How a policy does on robots other than the DROID Franka arm.",
     "sub": "PolaRiS supports only the DROID setup with joint-position control.",
     "basis": [
      "facts.robots"
     ]
    },
    {
     "id": "l3",
     "text": "How a policy handles soft objects or tasks with complex contact.",
     "sub": "All six tasks use rigid objects. The authors say the tasks leave out non-rigid objects.",
     "basis": [
      "facts.tasks",
      "issues.i4"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "PolaRiS paper (arXiv)",
     "date": "2025-12",
     "by": "authors",
     "method": "5 policies (pi0, pi0 at 100k steps, pi0-FAST, pi0.5 and PaliGemma-binning, so 4 models) were scored in simulation after 1k co-training steps, with 50 rollouts per task. They were also scored without co-training on the matching real scenes, with 20 rollouts per policy and scene. There were 6 scenes at UW and Princeton, scored by task progress.",
     "result": "Mean over 6 scenes: Pearson r = 0.90, MMRV = 0.03. MMRV measures how often two rankings disagree.",
     "authors_view": "strong",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Per-scene r 0.81 to 0.99 and MMRV 0.00 to 0.07 (Fig. 13); 0.90 is the plain mean of the six (our arithmetic). Each per-scene r rests on 5 points. All 5 policies come from Physical Intelligence's openpi DROID line."
    },
    {
     "id": "v2",
     "name": "PolaRiS paper (RSS 2026 version)",
     "date": "2026-07",
     "by": "authors",
     "method": "The same policies and scenes were used, with bootstrap resampling (repeated random sampling) of 50 episodes per policy, repeated 500 times.",
     "result": "Pearson r = 0.84 and MMRV = 0.08 in Fig. 6, with 95% intervals. The text gives r = 0.83.",
     "authors_view": "strong",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s3"
     ],
     "note": "Worst scene r 0.81 without bootstrapping. Under randomised controller gains (±10%) and object friction and mass (±20%), r stayed at 0.76 or above (Fig. 11). Intervals are drawn, not printed."
    },
    {
     "id": "v3",
     "name": "PolaRiS compared with RoboArena",
     "date": "2025-12",
     "by": "authors",
     "method": "The mean PolaRiS scores of 4 policies were compared with their mean progress scores in RoboArena's crowd-sourced real-robot evaluations. RoboArena uses different and broader tasks.",
     "result": "Pearson r = 0.98, MMRV = 0.00",
     "authors_view": "strong",
     "n_policies": 4,
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s17"
     ],
     "note": "RoboArena scores from its 2025-08-05 data dump (issue #5). pi0 at 100k steps excluded (not in RoboArena). RoboArena's authors include PolaRiS authors (Pertsch, Jain)."
    },
    {
     "id": "v4",
     "name": "PolaRiS paper: without co-training",
     "date": "2025-12",
     "by": "authors",
     "method": "The same policies were evaluated zero-shot, with no co-training in simulation, on 3 target scenes with 5 seeds.",
     "result": "Pearson r = 0.30 and MMRV = 0.29. With co-training, r = 0.86 and MMRV = 0.07.",
     "authors_view": "too low to accurately rank",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Fig. 10 (data-mix ablation), values read from the figure labels."
    },
    {
     "id": "v5",
     "name": "SimFoundry (NVIDIA, Georgia Tech, Stanford, UT Austin, Toronto)",
     "date": "2026-06",
     "by": "independent",
     "method": "5 policies (pi0, pi0.5, GR00T N1.6, GR00T N1.7 and DreamZero) were tested zero-shot on 4 tasks, and 3 fine-tuned policies on 3 tasks. The tests used DROID scenes rebuilt with PolaRiS's tools, without co-training in simulation. Each policy ran 25 rollouts per task, both on the real robot and in simulation.",
     "result": "Pearson r from -0.396 to 0.822 and MMRV from 0.053 to 0.352, depending on the task",
     "authors_view": "low correlation",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s20"
     ],
     "note": "Mean r about 0.31 over the 6 tasks where r is defined (our arithmetic). Not PolaRiS's own protocol (issues.i5). SimFoundry reports its own system at r 0.911, MMRV 0.018."
    }
   ],
   "searched": [
    {
     "for": "validity (paired sim-vs-real studies of PolaRiS)",
     "where": "PolaRiS arXiv v1, v2 and RSS 2026 PDF; SimFoundry 2606.28276; H2RBench 2609.24778 (built on PolaRiS, reports its own correlation); CoVer-VLA 2602.12281 (no real pairing on PolaRiS tasks); RoboWorld 2607.01060 (cites only); VLA Foundry 2604.19728 and IndustrialVLA-Bench 2609.25562 (no PolaRiS use found); PolaRiS GitHub issues; extended web searches.",
     "date": "2026-10-10"
    },
    {
     "for": "PolaRiS comparison with SimplerEnv",
     "where": "PolaRiS v2 and RSS text: no SimplerEnv run; it states SIMPLER cannot evaluate wrist-camera policies and cites OpenVLA for SIMPLER's weak correlation, but OpenVLA v1-v3 do not mention SIMPLER (arXiv 2406.09246).",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project site, README, Hub card.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "No single headline score; per-policy progress scores appear only in figures. Skipped.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (arXiv abstract page; v1 2025-12-18, v2 2025-12-30)",
     "url": "https://arxiv.org/abs/2512.16881",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "PolaRiS paper, full text v2 with figures (Figs. 6, 7, 8, 10, 13; Appendix C-D)",
     "url": "https://arxiv.org/html/2512.16881v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "PolaRiS, RSS 2026 proceedings PDF (revised numbers, bootstrap intervals, Figs. 5-6, 10-11)",
     "url": "https://www.roboticsproceedings.org/rss22/p062.pdf",
     "type": "paper",
     "publisher": "Robotics: Science and Systems XXII (Sydney, 13-17 July 2026)",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "PolaRiS, RSS 2026 proceedings page",
     "url": "https://www.roboticsproceedings.org/rss22/p062.html",
     "type": "paper",
     "publisher": "Robotics: Science and Systems XXII",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "PolaRiS project site",
     "url": "https://polaris-evals.github.io/",
     "type": "site",
     "publisher": "PolaRiS team",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: arhanjain/polaris (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/arhanjain/polaris",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "PolaRiS README (environments, checkpoints, cotraining)",
     "url": "https://github.com/arhanjain/polaris/blob/main/README.md",
     "type": "repo",
     "publisher": "Arhan Jain",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "PolaRiS LICENSE (MIT)",
     "url": "https://github.com/arhanjain/polaris/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Arhan Jain",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "PolaRiS pyproject.toml (version 0.1.0; isaaclab[all,isaacsim]==2.3.0)",
     "url": "https://github.com/arhanjain/polaris/blob/main/pyproject.toml",
     "type": "repo",
     "publisher": "Arhan Jain",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "PolaRiS commit history, tags and releases",
     "url": "https://github.com/arhanjain/polaris/commits/main",
     "type": "repo",
     "publisher": "Arhan Jain",
     "date": "2026-07-13",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "PolaRiS docs: checkpoints_and_envs.md",
     "url": "https://github.com/arhanjain/polaris/blob/main/docs/checkpoints_and_envs.md",
     "type": "repo",
     "publisher": "Arhan Jain",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "PolaRiS Hub dataset card (license: mit) and file tree",
     "url": "https://huggingface.co/datasets/owhan/PolaRiS-Hub",
     "type": "repo",
     "publisher": "PolaRiS team (owhan)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Hugging Face Hub API: owhan/PolaRiS-Hub (downloads, likes, commits, path history)",
     "url": "https://huggingface.co/api/datasets/owhan/PolaRiS-Hub?expand[]=downloads&expand[]=likes&expand[]=downloadsAllTime",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "PolaRiS co-training dataset card (license: mit) and dataset_info.json",
     "url": "https://huggingface.co/datasets/owhan/PolaRiS-datasets",
     "type": "repo",
     "publisher": "PolaRiS team (owhan)",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "PolaRiS issue #13: Poor Env Init Poses (owner replies)",
     "url": "https://github.com/arhanjain/polaris/issues/13",
     "type": "repo",
     "publisher": "arhanjain/polaris",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "PolaRiS issue #24: Reproducing the results (pi0.5) (owner reply 2026-06-24)",
     "url": "https://github.com/arhanjain/polaris/issues/24",
     "type": "repo",
     "publisher": "arhanjain/polaris",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "PolaRiS issue #5: roboarena results values (owner reply)",
     "url": "https://github.com/arhanjain/polaris/issues/5",
     "type": "repo",
     "publisher": "arhanjain/polaris",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "PolaRiS issue #8: Precise data in Figure 6",
     "url": "https://github.com/arhanjain/polaris/issues/8",
     "type": "repo",
     "publisher": "arhanjain/polaris",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Semantic Scholar API record for arXiv:2512.16881",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2512.16881?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "SimFoundry: Modular and Automated Scene Generation for Policy Learning and Evaluation (Section 5.1, Tables G.1-G.2, Appendix J)",
     "url": "https://arxiv.org/abs/2606.28276",
     "type": "paper",
     "publisher": "arXiv (NVIDIA, Georgia Tech, Stanford, UT Austin, Toronto)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Scaling Verification Can Be More Effective than Scaling Policy Learning for VLA Alignment (CoVer-VLA; Table 1 on PolaRiS)",
     "url": "https://arxiv.org/abs/2602.12281",
     "type": "paper",
     "publisher": "arXiv (Stanford, NVIDIA Research)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "H2RBench: A Real-to-Sim Benchmark for Evaluating Human-to-Robot Transfer (Section 3.3)",
     "url": "https://arxiv.org/abs/2609.24778",
     "type": "paper",
     "publisher": "arXiv (CMU, University of Michigan, TRI)",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "RLinf documentation: PolaRiS Evaluation",
     "url": "https://rlinf.readthedocs.io/en/latest/rst_source/evaluations/guides/polaris.html",
     "type": "site",
     "publisher": "RLinf",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "openpi: polaris_config.py (PolaRiS training configs, public checkpoint paths)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/src/openpi/training/misc/polaris_config.py",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Isaac Sim LICENSE (Apache-2.0 plus NVIDIA additional components)",
     "url": "https://github.com/isaac-sim/IsaacSim/blob/main/LICENSE",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "NVIDIA Isaac Sim Additional Software and Materials License",
     "url": "https://www.nvidia.com/en-us/agreements/enterprise-software/isaac-sim-additional-software-and-materials-license/",
     "type": "site",
     "publisher": "NVIDIA",
     "date": "2025-06-09",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources (no basic entry existed). Includes the RSS 2026 revisions, the SimFoundry independent check and the public-environment reproduction problems from the GitHub issues."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "procthor",
   "name": "ProcTHOR",
   "full_name": "ProcTHOR (ProcTHOR-10K, ArchitecTHOR)",
   "aliases": [
    "ProcTHOR",
    "ProcTHOR-10K",
    "ArchitecTHOR"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Procedural house generator and house dataset for embodied agents; ships ArchitecTHOR as a test-only ObjectNav set. Fits 'platform' with a dataset.",
   "summary": {
    "text": "Generator of procedurally built interactive houses in AI2-THOR, with a 10,000-house training set and a 10-house test set.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "PRIOR team, Allen Institute for AI",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2022-06 (arXiv v1 2022-06-14, only version)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "procthor release 0.0.1 on 2022-08-21; procthor-10k release 0.1.2 on 2022-07-11; last pushes 2023-04-07 (procthor) and 2022-12-14 (procthor-10k)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API. Checked 2026-10-10."
    },
    "published_at": {
     "value": "NeurIPS 2022 (Advances in Neural Information Processing Systems 35, Main Conference Track); NeurIPS virtual site labels it 'Outstanding Paper'",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Award label: https://neurips.cc/virtual/2022/poster/54832 Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "navigation",
      "mobile-manipulation"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Task families: navigation, interaction, manipulation; evaluated on ObjectNav, rearrangement, ArmPointNav. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-base",
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "ObjectNav trained with a simulated LoCoBot agent; arm agents for manipulation. Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Floorplans with 1-10 rooms; custom scene types such as classrooms, libraries, offices also possible. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "ArchitecTHOR ObjectNav reported as success rate and SPL (31.4% SR, 0.195 SPL). Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "None (of its own; results reported on other benchmarks' leaderboards (RoboTHOR, Habitat 2022, Rearrangement 2022)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. Checked 2026-10-10."
    },
    "license_data": {
     "value": "Apache-2.0 (procthor-10k LICENSE); paper datasheet: houses, asset database and code under Apache 2.0, ArchitecTHOR under Apache 2.0, usable for commercial and non-commercial purposes",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Datasheet sentence states commercial and non-commercial use. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "none-found",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "No real-robot experiments in the paper; datasheet lists Sim2Real only as a possible use."
    },
    "citations": {
     "value": 600,
     "display": "600",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 474,
     "display": "474",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "procthor repo; procthor-10k has 131. Checked 2026-10-10."
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "No releases since 2022, last push 2023-04. Checked 2026-10-10."
    },
    "version": {
     "value": "procthor package 0.0.1 (2022-08-21); ProcTHOR-10K 0.1.2 (2022-07-11)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "GitHub releases. Checked 2026-10-10."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "ProcTHOR: Large-Scale Embodied AI Using Procedural Generation",
     "url": "https://arxiv.org/abs/2206.06994",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-06"
    },
    "s2": {
     "title": "https://procthor.allenai.org/",
     "url": "https://procthor.allenai.org/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "allenai/procthor on GitHub (repository)",
     "url": "https://github.com/allenai/procthor",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "🏘️ ProcTHOR: Large-Scale Embodied AI Using Procedural Generation",
     "url": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/27c546ab1e4f1d7d638e6a8dfbad9a07-Abstract.html",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "ProcTHOR: Large-Scale Embodied AI Using Procedural Generation (full text)",
     "url": "https://arxiv.org/html/2206.06994",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2022-06"
    },
    "s6": {
     "title": "allenai/procthor-10k on GitHub (repository)",
     "url": "https://github.com/allenai/procthor-10k",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2206.06994",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "rbench",
   "name": "RBench",
   "aliases": [
    "RBench (ReVidgen)",
    "ReVidgen",
    "Rethinking Video Generation Model for the Embodied World"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Benchmark for video generators as robot world models: every case is a robot task or robot embodiment. Matches 'world-model evaluations built for robotics'.",
   "summary": {
    "text": "Scores video generators on robot task videos across five task types and four robot body types.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Peking University; ByteDance Seed (author affiliations 1 and 2)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2026-01 (arXiv v1 2026-01-21; repo created 2026-01-21; HF RBench dataset created 2026-01-15)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Repo/HF dates from APIs."
    },
    "latest_update": {
     "value": "Leaderboard data updated 2026-09-08 (commit 'Update leaderboard.json'); repo commit 2026-06-01 'Add Cosmos 3 RBench news'; RoVid-X released 2026-05-21.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Dates from HF Space commit API and GitHub API."
    },
    "version": {
     "value": "arXiv v1 only; published version in ICML 2026 (PMLR 306:24240-24283). No release tags.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "ICML 2026, PMLR vol. 306, pp. 24240-24283; also ICLR 2026 Workshop on World Models (OpenReview).",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Workshop venue from OpenReview API search (https://api2.openreview.net/notes/search)."
    },
    "venue": {
     "value": "offline",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Generated videos scored by MLLM judges and vision operators; no control loop."
    },
    "capability": {
     "value": [
      "world-modeling",
      "manipulation",
      "long-horizon",
      "collaboration"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Task categories include long-horizon planning and multi-entity collaboration."
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm",
      "humanoid",
      "legged"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "650 image-text pairs = 250 task-oriented (50 per category) + 400 embodiment-specific (100 per type).",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Same in paper Sec. 3.1."
    },
    "scoring": {
     "value": [
      "auto-judge",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "human_agreement": {
     "value": "Spearman rho = 0.96 (two-sided p < 1e-3) between RBench and human scores on a 10-model subset; 30 participants, pairwise A/B/tie votes.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Measured by the benchmark authors."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "Official (: HF Space, running on 2026-10-10; leaderboard.json has 30 entries.)"
    },
    "top_score": {
     "value": "WEAVE-0.5 0.642, LingBot-Video 0.620, Wan 2.6 0.607, Cosmos3-Nano 0.584 (avg, GPT-judged file)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The Space also has leaderboard_qwen.json (not opened; name suggests Qwen-judged scores)."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No LICENSE file (404) and GitHub reports no licence. README badge says 'Apache-2.0' but links to placeholder 'YOUR_LINK'."
    },
    "license_data": {
     "value": "CC-BY-4.0 (RBench evaluation set)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "HF dataset card metadata; not gated. Reference images come from public datasets and online sources; upstream terms not restated."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No link to robot execution. Human alignment only: Spearman rho = 0.96 (p < 1e-3) between RBench and pairwise human scores on 10 models, 30 participants, by the authors. README lists 'Embodied Execution Evaluation' via inverse dynamics model as a to-do, not done."
    },
    "citations": {
     "value": 33,
     "display": "33 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 101,
     "display": "101",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API."
    },
    "used_by": {
     "value": "NVIDIA Cosmos 3 report reports RBench I2V scores (Table 12; Cosmos3-Nano 58.4%).",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboard updated 2026-09-08."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Rethinking Video Generation Model for the Embodied World (full text)",
     "url": "https://arxiv.org/html/2601.15282v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s2": {
     "title": "Rethinking Video Generation Model for the Embodied World",
     "url": "https://arxiv.org/abs/2601.15282",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-01"
    },
    "s3": {
     "title": "DAGroup-PKU/RBench-Leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/DAGroup-PKU/RBench-Leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s4": {
     "title": "Rethinking Video Generation Model for the Embodied World",
     "url": "https://proceedings.mlr.press/v306/deng26r.html",
     "type": "paper",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "DAGroup-PKU/RBench on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/DAGroup-PKU/RBench",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "DAGroup-PKU/RBench-Leaderboard on Hugging Face (space)",
     "url": "https://huggingface.co/spaces/DAGroup-PKU/RBench-Leaderboard/blob/main/leaderboard.json",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s8": {
     "title": "DAGroup-PKU/ReVidgen on GitHub (repository)",
     "url": "https://github.com/DAGroup-PKU/ReVidgen",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "Cosmos 3: Omnimodal World Models for Physical AI (full text)",
     "url": "https://arxiv.org/html/2606.02800v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "rlbench",
   "name": "RLBench",
   "full_name": "RLBench: The Robot Learning Benchmark & Learning Environment",
   "aliases": [
    "RLBench-18 (PerAct protocol)",
    "RLBench-74 (Hiveformer protocol)",
    "FS10_V1 / FS95_V1 few-shot task sets"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "RLBench is a simulated benchmark of 100 hand-designed tasks for one Franka Panda arm, with demonstrations generated by a motion planner. Most papers report success rates on an 18-task subset defined by PerAct in 2022.",
    "short": "RLBench is a set of 100 simulated tasks for one Franka Panda robot arm. Papers on robot manipulation policies (the robot's control models) use it, and most report results on an 18-task subset.",
    "sources": [
     "s2",
     "s17"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents RLBench as a benchmark and learning environment with 100 tasks and a few-shot challenge."
    },
    "kind_secondary": {
     "value": [
      "platform"
     ],
     "display": "Also a learning environment with a task-building tool for adding new tasks",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "The paper describes a task builder and a public task repository."
    },
    "publishers": {
     "value": [
      "Imperial College London (Dyson Robotics Lab)"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Stephen James, Zicong Ma, David Rovick Arrojo, Andrew J. Davison (Dyson Robotics Lab and UROP, Imperial College London). The research was supported by Dyson Technology Ltd."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "University lab, industry-funded (Dyson)."
    },
    "region": {
     "value": "europe",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "United Kingdom."
    },
    "first_release": {
     "value": "2019-09",
     "display": "arXiv v1 on 2019-09-26 (repository created the same day). Published in IEEE Robotics and Automation Letters 5(2), pp. 3019-3026 (April 2020), presented at ICRA 2020.",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s5",
      "s7"
     ],
     "checked": "2026-10-11",
     "short": "September 2019, in IEEE Robotics and Automation Letters (RA-L) 2020"
    },
    "published_at": {
     "value": "IEEE Robotics and Automation Letters 5(2), 2020",
     "level": "verified",
     "sources": [
      "s3",
      "s5"
     ],
     "checked": "2026-10-11",
     "note": "DOI 10.1109/LRA.2020.2974707. README announcement 2020-01-28: accepted to RA-L with presentation at ICRA.",
     "short": "IEEE Robotics and Automation Letters, 2020"
    },
    "latest_update": {
     "value": "2025-01",
     "display": "Last commit 2025-01-25 (typo fix in a tutorial). Last functional changes 2024-07-02/03 (Gymnasium support, arm velocity and acceleration limits).",
     "level": "verified",
     "sources": [
      "s8",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "README announcements: v1.2.0 on 2022-02-18 (breaking action-mode API changes); shaped rewards for two tasks on 2022-05-11.",
     "short": "January 2025. The last commit fixed a typo."
    },
    "version": {
     "value": "1.2.0",
     "display": "1.2.0 in the code; the last GitHub release tag is 1.1.0 (2021-05-31). Not on PyPI; installed from GitHub.",
     "level": "verified",
     "sources": [
      "s10",
      "s9",
      "s5"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RLBench-18 (PerAct protocol)",
       "display": "18 tasks, some modified to add variations (249 variations in total), 100 training demos and 25 test episodes per task. Defined by PerAct (CoRL 2022), not by the RLBench authors. Used by most recent papers.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "RLBench-74 (Hiveformer protocol)",
       "display": "74 tasks, CoRL 2022.",
       "level": "verified",
       "sources": [
        "s29",
        "s32"
       ]
      },
      {
       "value": "Few-shot task sets",
       "display": "FS10_V1, FS25_V1, FS50_V1, FS95_V1: 10 to 95 training tasks plus 5 test tasks. Multi-task sets MT15_V1 to MT100_V1.",
       "level": "verified",
       "sources": [
        "s5",
        "s11"
       ],
       "note": "CONFLICT: the paper says 10% of the 100 tasks form the meta-test set; the README task sets hold out 5."
      }
     ],
     "short": "1.2.0. The last release tag is 1.1.0."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Tasks range from reaching to multi-stage tasks such as emptying a dishwasher. Each variation has text descriptions; instruction following became central through PerAct's language-conditioned protocol."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "In the common RLBench-18 protocol, test episodes use new object poses and sampled variations (colours, sizes, targets) that also appear in training. Train and test objects are the same.",
     "level": "verified",
     "sources": [
      "s17",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "PerAct states that it does not test generalisation to unseen objects. The original few-shot challenge (new tasks) is rarely reported; derived benchmarks such as GemBench and AGNOSTOS test new tasks.",
     "short": "New object positions. The objects are the same as in training."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "CoppeliaSim 4.1.0 (formerly V-REP) via PyRep",
     "level": "verified",
     "sources": [
      "s5",
      "s2",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "The README downloads CoppeliaSim Edu 4.1.0 for Ubuntu 20.04. The CoppeliaSim website lists V4.10.0 as current. PyRep is installed from GitHub; the PyPI name 'pyrep' belongs to an unrelated AGPL package.",
     "short": "CoppeliaSim 4.1, through PyRep"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda (simulated)",
     "level": "verified",
     "sources": [
      "s5",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Mico, Jaco, Sawyer and UR5 can be swapped in, but the README says the arm should remain the Franka Panda for benchmarking."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "display": "One constant scene: a Panda fixed to a wooden table, lit by 3 directional lights.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 100,
     "display": "100 tasks in the paper; 106 task files in the repository today",
     "level": "verified",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Repository count by us: .py files in rlbench/tasks excluding __init__.py (2026-10-10). Recent papers mostly use the 18-task subset.",
     "short": "100 tasks, of which 18 are commonly used"
    },
    "scenes": {
     "value": 1,
     "display": "1 shared table scene",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "1 scene"
    },
    "objects": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No object count in the paper, README or site. The README says models were supplied from turbosquid.com, cgtrader.com, free3d.com, thingiverse.com and cadnav.com."
    },
    "demonstrations": {
     "value": "generated on demand",
     "display": "Unlimited demonstrations generated by a motion planner (OMPL) from task waypoints; lengths 100 to 1,000 steps. No fixed official dataset.",
     "level": "verified",
     "sources": [
      "s2",
      "s18"
     ],
     "checked": "2026-10-11",
     "items": [
      {
       "value": "PerAct pre-generated data",
       "display": "Train 100, validation 25 and test 25 episodes per task for the 18 tasks, about 116 GB on Google Drive. PerAct says using them helps reproducibility because data generation samples scenes randomly.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      }
     ],
     "short": "Unlimited, generated by a motion planner"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Each task has a sparse reward of +1 on completion, checked by task-specific success conditions."
    },
    "metric_detail": {
     "value": "Average task success rate (%)",
     "display": "In the RLBench-18 protocol: one multi-task policy, each test episode scored 0 or 100, averaged over 25 episodes per task and over the 18 tasks. The paper's own challenge asks for 1-, 5- and 20-shot success on held-out tasks.",
     "level": "verified",
     "sources": [
      "s17",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "Success rate over 18 tasks"
    },
    "trials": {
     "value": "25 test episodes per task (RLBench-18)",
     "level": "verified",
     "sources": [
      "s17",
      "s19",
      "s28"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "PerAct",
       "display": "25 episodes per task (450 in total), at most 25 keyframe steps; best checkpoint chosen on a separate 25-episode validation set.",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "RVT and later",
       "display": "Each model run 5 times on the same 25 episodes per task, because the sampling-based motion planner is random; mean and standard deviation reported.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "BridgeVLA++",
       "display": "Mean ± std over five random seeds, 25 evaluation episodes per seed.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      }
     ],
     "short": "25 per task, often repeated in 5 runs"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s17",
      "s19",
      "s25",
      "s28",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "RVT, RVT-2, 3D Diffuser Actor, SAM2Act, BridgeVLA and BridgeVLA++ report standard deviations over 5 runs. PerAct reports one evaluation; COLOSSEUM reports one training seed and one evaluation seed."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s6",
      "s5",
      "s15"
     ],
     "checked": "2026-10-11",
     "note": "No submission process or organiser-run evaluation on the site or README."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s6",
      "s5",
      "s15"
     ],
     "checked": "2026-10-11",
     "note": "No leaderboard on the project site or README. In issue #171 (2022) the maintainer said no official baselines exist because simple policies did not work on these sparse-reward tasks."
    },
    "top_score": {
     "value": 93.7,
     "display": "93.7% average success on the 18-task protocol (BridgeVLA++, August 2026). The highest RLBench-18 result we found. Papers differ in inputs and training pipelines.",
     "level": "inferred",
     "sources": [
      "s28",
      "s27"
     ],
     "checked": "2026-10-10",
     "note": "Each score is verified at its source. 'Highest' is our judgement from two web searches on 2026-10-10 and the papers we opened; there is no tracker or leaderboard for RLBench, and the shared search budget ran out before a fuller search. All rows: RLBench-18, 100 demonstrations per task, success rate in %.",
     "items": [
      {
       "value": 62.9,
       "display": "RVT (NVIDIA), 2023-06: 62.9% average over 18 tasks",
       "level": "verified",
       "sources": [
        "s19"
       ],
       "note": "RVT re-evaluated PerAct's released model at 49.4%. PerAct's own paper gives per-task numbers only.",
       "data": {
        "model": "RVT",
        "date": "2023-06",
        "avg": 62.9,
        "rl": false
       }
      },
      {
       "value": 65,
       "display": "Act3D, 2023-06: 65%",
       "level": "verified",
       "sources": [
        "s20"
       ],
       "data": {
        "model": "Act3D",
        "date": "2023-06",
        "avg": 65,
        "rl": false
       }
      },
      {
       "value": 81.3,
       "display": "3D Diffuser Actor, 2024-02: 81.3%",
       "level": "verified",
       "sources": [
        "s21"
       ],
       "data": {
        "model": "3D Diffuser Actor",
        "date": "2024-02",
        "avg": 81.3,
        "rl": false
       }
      },
      {
       "value": 70.6,
       "display": "SAM-E, 2024-05: 70.6 ± 0.7%",
       "level": "verified",
       "sources": [
        "s22",
        "s25"
       ],
       "data": {
        "model": "SAM-E",
        "date": "2024-05",
        "avg": 70.6,
        "rl": false
       }
      },
      {
       "value": 81.4,
       "display": "RVT-2 (NVIDIA), 2024-06: 81.4%",
       "level": "verified",
       "sources": [
        "s23"
       ],
       "note": "ARP found 77.0% without the data loader's randomised timestep and 74.1% with correct timesteps (see issues.i2).",
       "data": {
        "model": "RVT-2",
        "date": "2024-06",
        "avg": 81.4,
        "rl": false
       }
      },
      {
       "value": 84.9,
       "display": "ARP+, 2024-10: 84.9% (ARP: 81.6%)",
       "level": "verified",
       "sources": [
        "s24"
       ],
       "data": {
        "model": "ARP+",
        "date": "2024-10",
        "avg": 84.9,
        "rl": false
       }
      },
      {
       "value": 86.8,
       "display": "SAM2Act, 2025-01: 86.8 ± 0.5%",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "data": {
        "model": "SAM2Act",
        "date": "2025-01",
        "avg": 86.8,
        "rl": false
       }
      },
      {
       "value": 88.2,
       "display": "BridgeVLA (CASIA, ByteDance Seed), 2025-06: 88.2%",
       "level": "verified",
       "sources": [
        "s26",
        "s28"
       ],
       "note": "BridgeVLA++ later reports BridgeVLA at 90.5 ± 1.1% and labels 88.2% as the discretised-rotation version.",
       "data": {
        "model": "BridgeVLA",
        "date": "2025-06",
        "avg": 88.2,
        "rl": false
       }
      },
      {
       "value": 91.8,
       "display": "ActiveVLA, 2026-01: 91.8%, average rank 1.22",
       "level": "verified",
       "sources": [
        "s27"
       ],
       "note": "CVPR 2026.",
       "data": {
        "model": "ActiveVLA",
        "date": "2026-01",
        "avg": 91.8,
        "rl": false
       }
      },
      {
       "value": 93.7,
       "display": "BridgeVLA++, 2026-08: 93.7 ± 0.6% (five seeds, 25 episodes per seed)",
       "level": "verified",
       "sources": [
        "s28"
       ],
       "data": {
        "model": "BridgeVLA++",
        "date": "2026-08",
        "avg": 93.7,
        "rl": false
       }
      }
     ],
     "short": "93.7% on 18 tasks (August 2026)"
    },
    "license_code": {
     "value": "Custom: Imperial College London RLBench Licence Agreement (non-commercial)",
     "display": "Custom licence: free, non-exclusive, non-transferable; use only for non-commercial, internal or academic research. BSD terms for BSD elements.",
     "level": "verified",
     "sources": [
      "s4",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Key clauses: 2(b) bans commercial use, including research to develop products for sale and paid services; commercial use requires contacting researchcontracts.engineering@imperial.ac.uk. 2(d) bans transfer and sub-licensing. 2(f) requires publications to credit the software as licensed from Imperial and to send Imperial a copy. 7(a) Imperial may terminate at any time; 7(d) restrictions expire 10 years after first use. English law. GitHub shows 'NOASSERTION'.",
     "short": "Custom licence, non-commercial use only"
    },
    "license_data": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2",
      "s18"
     ],
     "checked": "2026-10-11",
     "note": "No official dataset is distributed; users generate demonstrations. PerAct's pre-generated data are produced with RLBench and hosted by the PerAct authors."
    },
    "license_assets": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The README says 3D models came from turbosquid.com, cgtrader.com, free3d.com, thingiverse.com and cadnav.com. No per-model licence or attribution file was found in the README or LICENSE; the RLBench licence covers 'the Software'."
    },
    "simulator_licence": {
     "value": "CoppeliaSim Edu: education only",
     "display": "The README installs CoppeliaSim Edu. Its terms allow use only by students, teachers and professors of schools and universities, and bar companies, research institutions and commercial use. The Pro edition allows commercial use.",
     "level": "verified",
     "sources": [
      "s12",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "The Edu edition is for schools and universities only."
    },
    "access": {
     "value": "open",
     "display": "Code on GitHub with no registration; downloading binds the user to the licence. The simulator download requires accepting CoppeliaSim's terms.",
     "level": "verified",
     "sources": [
      "s5",
      "s4",
      "s12"
     ],
     "checked": "2026-10-10",
     "short": "Open. Downloading the code binds the user to the licence."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s4",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "From clause 2(b) of the RLBench licence, plus the CoppeliaSim Edu terms. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Many policies scored on RLBench also ran on real robots (PerAct, RVT-2, BridgeVLA and others), on different real tasks. No study compares the same policies' RLBench and real-robot scores.",
     "level": "inferred",
     "sources": [
      "s2",
      "s17",
      "s23",
      "s26",
      "s30",
      "s39",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "The RLBench paper offers domain randomisation and swappable arms for sim-to-real research but has no real-robot experiment. The closest evidence is THE COLOSSEUM (RSS 2024), built on RLBench: it mirrored 4 tasks with 3D-printed objects on a real Franka, trained one PerAct on 5 real demonstrations per task and another in simulation, and tested each under matching perturbations (10 episodes × 3 runs). Per-factor R² between the two ranged from 0.46 to 0.94 (mean 0.614). These are two separately trained models of one method compared across perturbation conditions, so we do not count it as a paired evaluation of the same policies. PolaRiS and the 2026 sim-real recipe paper mention RLBench only in related work.",
     "short": "No study has compared its scores with real-robot scores for the same policies."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "THE COLOSSEUM",
      "PerAct2 bimanual tasks",
      "GemBench",
      "VLMbench",
      "MemoryBench",
      "AGNOSTOS"
     ],
     "display": "Benchmarks built on RLBench tasks or its simulator",
     "level": "verified",
     "sources": [
      "s30",
      "s31",
      "s32",
      "s33",
      "s25",
      "s34",
      "s35"
     ],
     "checked": "2026-10-11",
     "note": "Colosseum V2 (2026) shares the name but is built on ManiSkill, not RLBench.",
     "items": [
      {
       "value": "THE COLOSSEUM",
       "display": "2024, RSS 2024. 20 tasks, 14 perturbation factors, a real-robot mirror of 4 tasks. 18 papers reported results by 2026-05 per the 2026 audit's tracker.",
       "level": "verified",
       "sources": [
        "s30",
        "s37"
       ]
      },
      {
       "value": "PerAct2 bimanual tasks",
       "display": "2024. Extends RLBench to two arms: 13 new tasks with 23 variations.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "GemBench",
       "display": "2024, ICRA 2025. 16 training and 44 test tasks at four generalisation levels.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "VLMbench",
       "display": "2022. Compositional language tasks; code based on RLBench under the RLBench licence.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "MemoryBench",
       "display": "2025 (SAM2Act paper). Extends the RLBench simulator with 3 spatial-memory tasks.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "AGNOSTOS",
       "display": "2025. 23 unseen tasks for cross-task generalisation, built on RLBench.",
       "level": "verified",
       "sources": [
        "s34"
       ]
      }
     ],
     "short": "6 benchmarks built on RLBench"
    },
    "citations": {
     "value": 1139,
     "display": "1,139 (Semantic Scholar; 155 influential)",
     "level": "verified",
     "sources": [
      "s38"
     ],
     "checked": "2026-10-10",
     "short": "1,139"
    },
    "github_stars": {
     "value": 1826,
     "display": "1,826 stars, 324 forks, 94 open issues (stepjam/RLBench)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "1,826"
    },
    "used_by": {
     "value": "RLBench-18 results are reported by many 3D manipulation policy papers from 2022 to 2026; we found no count of papers.",
     "level": "verified",
     "sources": [
      "s17",
      "s19",
      "s23",
      "s25",
      "s26",
      "s27",
      "s28",
      "s36"
     ],
     "checked": "2026-10-10",
     "note": "The 2026 audit did not audit RLBench; it focused on the five benchmarks it found most reported in recent VLA work.",
     "items": [
      {
       "value": "PerAct",
       "display": "University of Washington and NVIDIA, 2022-09; defined the RLBench-18 protocol",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "RVT, RVT-2",
       "display": "NVIDIA, 2023-2024",
       "level": "verified",
       "sources": [
        "s19",
        "s23"
       ]
      },
      {
       "value": "Act3D, 3D Diffuser Actor",
       "display": "Carnegie Mellon University, 2023-2024",
       "level": "verified",
       "sources": [
        "s20",
        "s21"
       ]
      },
      {
       "value": "SAM2Act",
       "display": "University of Washington, NVIDIA, Allen Institute for AI, 2025-01",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "BridgeVLA, BridgeVLA++",
       "display": "CASIA and ByteDance Seed, 2025-2026",
       "level": "verified",
       "sources": [
        "s26",
        "s28"
       ]
      },
      {
       "value": "ActiveVLA",
       "display": "Fudan University and others, 2026-01",
       "level": "verified",
       "sources": [
        "s27"
       ]
      }
     ],
     "short": "Mainly papers on 3D manipulation policies"
    },
    "industry_use": {
     "value": [
      "NVIDIA",
      "ByteDance",
      "Dyson"
     ],
     "level": "verified",
     "sources": [
      "s19",
      "s23",
      "s26",
      "s28",
      "s2"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "NVIDIA",
       "display": "RVT and RVT-2 are NVIDIA papers; NVIDIA authors also co-wrote PerAct, COLOSSEUM and SAM2Act.",
       "level": "verified",
       "sources": [
        "s19",
        "s23",
        "s17",
        "s30",
        "s25"
       ]
      },
      {
       "value": "ByteDance",
       "display": "BridgeVLA lists ByteDance Seed; BridgeVLA++ authors note work done at ByteDance Seed.",
       "level": "verified",
       "sources": [
        "s26",
        "s28"
       ]
      },
      {
       "value": "Dyson",
       "display": "Funded the original RLBench work.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "status": {
     "value": "dormant",
     "display": "No functional change since July 2024; last commit January 2025. Still used in new papers.",
     "level": "inferred",
     "sources": [
      "s8",
      "s7",
      "s16"
     ],
     "checked": "2026-10-11",
     "note": "94 open issues. Issue #268 (2025-01, open): the bulb holder is missing from generated light_bulb_in demonstrations; the reporter switched tasks.",
     "short": "No functional changes since July 2024. Still used in new papers."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "Most RLBench results come from a modified 18-task subset",
     "text": "Most recent papers report PerAct's RLBench-18 protocol: 18 tasks, some modified to add variations (249 in total), 100 demonstrations and 25 test episodes per task, four 128 × 128 RGB-D cameras and keyframe actions. Others use 74 tasks (Hiveformer) or their own subsets. The paper's own few-shot challenge is rarely reported, and the paper (10% of tasks held out) and README (5 test tasks) disagree on its split.",
     "level": "verified",
     "sources": [
      "s17",
      "s29",
      "s2",
      "s5"
     ],
     "status": "open",
     "short": "Most papers report results on a modified subset of 18 tasks defined by PerAct, instead of the full 100-task benchmark."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same model gets different scores",
     "text": "PerAct's paper gives no average; RVT re-evaluated its released model at 49.4%. BridgeVLA reports 88.2%; its follow-up BridgeVLA++ reports BridgeVLA at 90.5 ± 1.1%. ARP found that RVT-2's 81.4% depends on a data-loader detail inherited from C2F-ARM: removing a randomised timestep gives 77.0%, and giving correct timesteps gives 74.1%. RVT notes that earlier baselines picked the best model per task on validation sets, which overstates their multi-task scores.",
     "level": "verified",
     "sources": [
      "s19",
      "s26",
      "s28",
      "s24",
      "s17"
     ],
     "status": "open",
     "short": "The score of one model, RVT-2, falls from 81.4% to 74.1% after one change to how training data is loaded."
    },
    {
     "id": "i3",
     "type": "saturated",
     "title": "Top scores are approaching 100%",
     "text": "RLBench-18 averages rose from 62.9% (RVT, 2023-06) to 91.8% (ActiveVLA, 2026-01) and 93.7% (BridgeVLA++, 2026-08). Many individual tasks are at or near 100%; in BridgeVLA++ the place cups task is at 76.8%.",
     "level": "verified",
     "sources": [
      "s19",
      "s27",
      "s28"
     ],
     "status": "open",
     "short": "The best average over the 18 tasks is 93.7%."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Scores fall when the scene changes",
     "text": "THE COLOSSEUM perturbs 20 RLBench tasks along 14 factors such as colour, texture, lighting, distractors and camera pose. It reports success dropping 30 to 50% per factor and at least 75% when factors are combined. PerAct trained without perturbations fell from 34.5% to 6.4% with all perturbations on.",
     "level": "verified",
     "sources": [
      "s30"
     ],
     "status": "open",
     "short": "In THE COLOSSEUM, success drops by 30 to 50% for each type of scene change, and by at least 75% when changes are combined."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "The licence is non-commercial and the simulator edition is for education only",
     "text": "RLBench's custom licence allows only non-commercial, internal or academic research, bans transfer and sub-licensing, lets Imperial terminate at any time, and asks publications to credit Imperial and send it a copy. The README installs CoppeliaSim Edu, whose terms exclude companies, research institutions and commercial use. Main forks keep the custom licence (GitHub shows NOASSERTION).",
     "level": "verified",
     "sources": [
      "s4",
      "s12",
      "s5",
      "s14"
     ],
     "status": "open",
     "note": "Not legal advice.",
     "short": "The licence does not allow commercial use. The default simulator edition is for schools and universities only."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Maintenance has stopped and 3D assets lack licences",
     "text": "The last functional commit is from July 2024 and the code pins CoppeliaSim 4.1.0 (the vendor's current version is 4.10.0). 94 issues are open. The README credits 3D models to five third-party model sites without per-model licence information.",
     "level": "verified",
     "sources": [
      "s8",
      "s5",
      "s12",
      "s7"
     ],
     "status": "open",
     "short": "The code has had no functional changes since July 2024 and uses an old simulator version. The third-party 3D models have no licence information."
    },
    {
     "id": "i7",
     "type": "protocol-variance",
     "title": "Results vary from one evaluation run to the next",
     "text": "RVT runs each model five times on the same 25 episodes per task because the sampling-based motion planner is random. PerAct distributes fixed pre-generated datasets because data generation samples scenes randomly. Some papers report one evaluation run.",
     "level": "verified",
     "sources": [
      "s19",
      "s18",
      "s30"
     ],
     "status": "open",
     "short": "The motion planner makes random choices, so results vary between runs. Some papers report only one run."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "When a paper reports 'RLBench', it almost always means PerAct's 18-task protocol, not the 100-task benchmark or its few-shot challenge. Check the task subset, the number of demonstrations and the inputs before comparing.",
     "basis": [
      "issues.i1",
      "facts.version"
     ],
     "confidence": "high",
     "short": "Check which RLBench subset and protocol a number uses."
    },
    {
     "id": "r2",
     "text": "A high RLBench-18 score shows a policy fits 18 known tasks with seen objects in one scene. Scene perturbations cut scores sharply, and no study has checked RLBench scores against real-robot results for the same policies.",
     "basis": [
      "facts.generalisation",
      "issues.i4",
      "facts.sim_to_real"
     ],
     "confidence": "medium",
     "short": "A high score shows that a policy fits 18 known tasks. It does not show that the policy copes with scene changes."
    },
    {
     "id": "r3",
     "text": "Gaps of a few points at the top are not reliable. Data-loader details move the same model by several points, and papers re-report earlier models with different numbers.",
     "basis": [
      "issues.i2",
      "issues.i7",
      "issues.i3"
     ],
     "confidence": "high",
     "short": "Ignore gaps of a few points between top results."
    },
    {
     "id": "r4",
     "text": "Companies should read the licence before using RLBench. It and the default simulator edition exclude commercial use and product research without permission. This is our reading, not legal advice.",
     "basis": [
      "facts.license_code",
      "facts.simulator_licence",
      "issues.i5"
     ],
     "confidence": "high",
     "short": "Commercial users need permission from Imperial College London. The default edition of the simulator also excludes commercial use."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy will do on a real robot.",
     "sub": "No study has compared RLBench and real-robot scores for the same policies.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy handles new objects.",
     "sub": "The common protocol tests only objects seen in training.",
     "basis": [
      "facts.generalisation"
     ]
    },
    {
     "id": "l3",
     "text": "How well a policy copes with changes to the scene.",
     "sub": "THE COLOSSEUM, a test built on RLBench, shows large drops in success under scene changes.",
     "basis": [
      "issues.i4"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "RLBench paper (no real-robot experiment); README and site; PerAct, RVT-2, BridgeVLA (real results on separate tasks); THE COLOSSEUM (perturbation comparison, one method); PolaRiS and 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (related work only). The shared web-search budget ran out before a dedicated search for RLBench sim-real studies.",
     "date": "2026-10-11"
    },
    {
     "for": "leaderboard",
     "where": "Project site, README, issue #171, two web searches for RLBench-18 results (2026-10-10).",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "README acknowledgements, LICENSE, repository root.",
     "date": "2026-10-10"
    },
    {
     "for": "prior claim 'forks mislabel MIT'",
     "where": "GitHub API licence fields for stepjam/RLBench, MohitShridhar/RLBench (PerAct fork), markusgrotz/RLBench (PerAct2 fork); PyPI for 'rlbench' and 'pyrep'.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RLBench: The Robot Learning Benchmark & Learning Environment (arXiv abstract page)",
     "url": "https://arxiv.org/abs/1909.12271",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2019-09",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RLBench paper, full text v1",
     "url": "https://arxiv.org/pdf/1909.12271",
     "type": "paper",
     "publisher": "arXiv (Dyson Robotics Lab, Imperial College London)",
     "date": "2019-09",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Crossref record for DOI 10.1109/LRA.2020.2974707 (RA-L vol. 5, no. 2, pp. 3019-3026)",
     "url": "https://api.crossref.org/works/10.1109/lra.2020.2974707",
     "type": "index",
     "publisher": "Crossref",
     "date": "2020-04",
     "accessed": "2026-10-11"
    },
    "s4": {
     "title": "RLBench LICENSE file (Imperial College London RLBench Licence Agreement)",
     "url": "https://github.com/stepjam/RLBench/blob/master/LICENSE",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2019",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RLBench GitHub README (install, task sets, announcements, acknowledgements)",
     "url": "https://github.com/stepjam/RLBench/blob/master/README.md",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RLBench project site",
     "url": "https://sites.google.com/view/rlbench",
     "type": "site",
     "publisher": "Dyson Robotics Lab, Imperial College London",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: stepjam/RLBench (stars, forks, open issues, licence 'NOASSERTION')",
     "url": "https://api.github.com/repos/stepjam/RLBench",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "RLBench commit history",
     "url": "https://github.com/stepjam/RLBench/commits/master",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2025-01-25",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "RLBench releases (latest tag 1.1.0, 2021-05-31)",
     "url": "https://github.com/stepjam/RLBench/releases",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2021-05-31",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "rlbench/__init__.py (__version__ = '1.2.0')",
     "url": "https://github.com/stepjam/RLBench/blob/master/rlbench/__init__.py",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "rlbench/tasks directory and tasks/__init__.py (task files; FS and MT task sets)",
     "url": "https://github.com/stepjam/RLBench/tree/master/rlbench/tasks",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "CoppeliaSim website (editions and Edu terms)",
     "url": "https://www.coppeliarobotics.com/",
     "type": "site",
     "publisher": "Coppelia Robotics",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "PyRep repository (MIT licence)",
     "url": "https://github.com/stepjam/PyRep",
     "type": "repo",
     "publisher": "stepjam/PyRep",
     "date": "2024-08",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "GitHub API: forks MohitShridhar/RLBench and markusgrotz/RLBench (licence field)",
     "url": "https://api.github.com/repos/MohitShridhar/RLBench",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "RLBench issue #171: Baseline Imitation Learning Policies & Results",
     "url": "https://github.com/stepjam/RLBench/issues/171",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2022-06",
     "accessed": "2026-10-11"
    },
    "s16": {
     "title": "RLBench issue #268: Bulb holder not visible in demonstrations generated for light_bulb_in (open)",
     "url": "https://github.com/stepjam/RLBench/issues/268",
     "type": "repo",
     "publisher": "stepjam/RLBench",
     "date": "2025-01",
     "accessed": "2026-10-11"
    },
    "s17": {
     "title": "Perceiver-Actor: A Multi-Task Transformer for Robotic Manipulation (PerAct)",
     "url": "https://arxiv.org/abs/2209.05451",
     "type": "paper",
     "publisher": "CoRL 2022 (University of Washington, NVIDIA)",
     "date": "2022-09",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "PerAct repository README (pre-generated RLBench datasets, licences)",
     "url": "https://github.com/peract/peract",
     "type": "repo",
     "publisher": "peract/peract",
     "date": "2024-05",
     "accessed": "2026-10-11"
    },
    "s19": {
     "title": "RVT: Robotic View Transformer for 3D Object Manipulation",
     "url": "https://arxiv.org/abs/2306.14896",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Act3D: 3D Feature Field Transformers for Multi-Task Robotic Manipulation",
     "url": "https://arxiv.org/abs/2306.17817",
     "type": "paper",
     "publisher": "arXiv (Carnegie Mellon University)",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "3D Diffuser Actor: Policy Diffusion with 3D Scene Representations, v1",
     "url": "https://arxiv.org/pdf/2402.10885v1",
     "type": "paper",
     "publisher": "arXiv (Carnegie Mellon University)",
     "date": "2024-02",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "SAM-E: Leveraging Visual Foundation Model with Sequence Imitation for Embodied Manipulation",
     "url": "https://arxiv.org/abs/2405.19586",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "RVT-2: Learning Precise Manipulation from Few Demonstrations",
     "url": "https://arxiv.org/abs/2406.08545",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Autoregressive Action Sequence Learning for Robotic Manipulation (ARP; RA-L 2025, Appendix Table A4)",
     "url": "https://arxiv.org/abs/2410.03132",
     "type": "paper",
     "publisher": "IEEE RA-L (Rutgers University)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "SAM2Act: Integrating Visual Foundation Model with A Memory Architecture for Robotic Manipulation",
     "url": "https://arxiv.org/abs/2501.18564",
     "type": "paper",
     "publisher": "arXiv (University of Washington, Universidad Católica San Pablo, NVIDIA, Allen Institute for AI)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "BridgeVLA: Input-Output Alignment for Efficient 3D Manipulation Learning with Vision-Language Models",
     "url": "https://arxiv.org/abs/2506.07961",
     "type": "paper",
     "publisher": "arXiv (CASIA, ByteDance Seed and others)",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "ActiveVLA: Injecting Active Perception into Vision-Language-Action Models for Precise 3D Robotic Manipulation",
     "url": "https://arxiv.org/abs/2601.08325",
     "type": "paper",
     "publisher": "CVPR 2026 (Fudan University, Shanghai Innovation Institute, NTU)",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "BridgeVLA++: A Data-Efficient, Generalizable, and Memory-Augmented Vision-Language-Action Framework for 3D Manipulation",
     "url": "https://arxiv.org/abs/2608.05042",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Instruction-driven history-aware policies for robotic manipulations (Hiveformer; 74 RLBench tasks)",
     "url": "https://arxiv.org/abs/2209.04899",
     "type": "paper",
     "publisher": "CoRL 2022",
     "date": "2022-09",
     "accessed": "2026-10-11"
    },
    "s30": {
     "title": "THE COLOSSEUM: A Benchmark for Evaluating Generalization for Robotic Manipulation (v2)",
     "url": "https://arxiv.org/abs/2402.08191",
     "type": "paper",
     "publisher": "RSS 2024 (Universidad Católica San Pablo, USC, UW, AI2, NVIDIA)",
     "date": "2024-02",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "PerAct2: Benchmarking and Learning for Robotic Bimanual Manipulation Tasks",
     "url": "https://arxiv.org/abs/2407.00278",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-07",
     "accessed": "2026-10-11"
    },
    "s32": {
     "title": "Towards Generalizable Vision-Language Robotic Manipulation: A Benchmark and LLM-guided 3D Policy (GemBench)",
     "url": "https://arxiv.org/abs/2410.01345",
     "type": "paper",
     "publisher": "ICRA 2025",
     "date": "2024-10",
     "accessed": "2026-10-11"
    },
    "s33": {
     "title": "VLMbench: A Compositional Benchmark for Vision-and-Language Manipulation",
     "url": "https://arxiv.org/abs/2206.08522",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2022-06",
     "accessed": "2026-10-11"
    },
    "s34": {
     "title": "Exploring the Limits of Vision-Language-Action Manipulations in Cross-task Generalization (AGNOSTOS)",
     "url": "https://arxiv.org/abs/2505.15660",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-11"
    },
    "s35": {
     "title": "Colosseum V2: Benchmarking Generalization for Vision-Language-Action Models (built on ManiSkill)",
     "url": "https://arxiv.org/abs/2605.27759",
     "type": "paper",
     "publisher": "IEEE RA-L (accepted, per arXiv comment)",
     "date": "2026-05",
     "accessed": "2026-10-11"
    },
    "s36": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (2026 audit)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Audit release: COLOSSEUM citation tracker (the_colossem.csv)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/leaderboards/others/the_colossem.csv",
     "type": "repo",
     "publisher": "ripl/ManipulationBenchmarkAudit",
     "date": "2026-05",
     "accessed": "2026-10-11"
    },
    "s38": {
     "title": "Semantic Scholar batch API record for ARXIV:1909.12271 (also 2209.05451 and 2402.08191)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies",
     "url": "https://arxiv.org/abs/2512.16881",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation",
     "url": "https://arxiv.org/abs/2606.10366",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from the checked basic entry and primary sources. Added the RLBench-18 score series, licence clause details, the CoppeliaSim Edu terms, derived benchmarks and reporting issues. Checking continued into 2026-10-11 (local time); sources opened then carry that access date."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "roboarena",
   "name": "RoboArena",
   "full_name": "RoboArena: Distributed Real-World Evaluation of Generalist Robot Policies",
   "aliases": [
    "DROID-RoboArena",
    "RoboArena x CoRL 2025 Challenge (related event)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "RoboArena is a real-robot evaluation service for generalist robot policies on the DROID Franka arm. Volunteer evaluators at many sites run blind A/B tests of two policies on tasks they choose, and the preferences are combined into a rating.",
    "short": "RoboArena is a service that tests robot policies (the robots' control models) on real DROID robot arms. Volunteers run blind A/B tests, in which two unnamed policies try the same task, and their preferences are combined into a rating.",
    "sources": [
     "s1",
     "s2",
     "s5"
    ]
   },
   "facts": {
    "kind": {
     "value": "arena",
     "level": "inferred",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas. Submitters host a policy server; RoboArena's evaluator network runs it on real robots. RobotArena (with the infinity sign) is a different project."
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "Also publishes its evaluation data (videos, robot actions, instructions, preferences and progress scores) as Hugging Face data dumps.",
     "level": "verified",
     "sources": [
      "s20",
      "s21"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "live service, no version numbers",
     "display": "No versions, tags or releases. The leaderboard is recomputed continuously and three data dumps exist (2025-08-05, 2026-02-03, 2026-07-17).",
     "level": "verified",
     "sources": [
      "s17",
      "s20",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper v1 (2025-06-22) and v2 (2025-11-29)",
       "display": "v2 adds Table 2 (simulated task and policy drift) and two authors. Figure 6 is identical in both.",
       "level": "verified",
       "sources": [
        "s1",
        "s2",
        "s4"
       ]
      },
      {
       "value": "Official and All views (since 2026-06-10)",
       "display": "The official view lists only policies with at least 100 A/B evaluations. The all-policies view adds entries with fewer, marked 'low sample'.",
       "level": "verified",
       "sources": [
        "s5",
        "s15"
       ]
      },
      {
       "value": "Integrity update (2026-06-11 to 2026-06-13)",
       "display": "New evaluator rule, evaluations from 2026-04-02 rolled back, audit metrics added. See issues.i1.",
       "level": "verified",
       "sources": [
        "s14",
        "s15",
        "s16"
       ]
      }
     ],
     "short": "No version numbers. The leaderboard is live, and there are 3 data dumps."
    },
    "publishers": {
     "value": [
      "UC Berkeley",
      "Stanford University",
      "University of Washington",
      "University of Montreal",
      "NVIDIA",
      "University of Pennsylvania",
      "UT Austin",
      "Yonsei University"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "UC Berkeley",
       "display": "Lead institution; two of three corresponding authors (Atreya, Pertsch).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Stanford University",
       "display": "Third corresponding author (Tony Lee).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "University of Washington, University of Montreal, University of Pennsylvania, UT Austin, Yonsei University",
       "display": "Evaluator sites and co-authors in the paper.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Co-authors who built the simulated evaluation environments (Appendix A).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "note": "arXiv v2 lists 32 authors; the PMLR version lists 26. Eight numbered affiliations in arXiv v2."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Seven of eight affiliations are universities; one is NVIDIA."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Lead institutions are US universities. Evaluator labels also include sites in Canada, South Korea and the Czech Republic."
    },
    "first_release": {
     "value": "2025-06",
     "display": "arXiv v1 on 2025-06-22. Published at CoRL 2025 (PMLR volume 305). The first public A/B evaluation is dated 2025-04-15.",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Code repository created 2025-06-14; website repository created 2025-06-17.",
     "short": "June 2025, at CoRL 2025"
    },
    "latest_update": {
     "value": "2026-10",
     "display": "Leaderboard recomputed 2026-10-11 03:30 UTC (evening of 2026-10-10 in the US). Newest public A/B evaluation 2026-09-12. Latest data dump 2026-07-17. Code last changed 2026-04-28.",
     "level": "verified",
     "sources": [
      "s6",
      "s8",
      "s21",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "The leaderboard was recomputed in October 2026. The newest evaluation is from 12 September 2026."
    },
    "status": {
     "value": "active",
     "display": "Evaluations every month from April to September 2026, at a lower rate than in 2025. The site says the benchmark runs live through December 2026, with possible extensions.",
     "level": "inferred",
     "sources": [
      "s9",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Counted evaluations per month in 2026 (our count from the public list): Jan 234, Feb 237, Mar 171, Apr 74, May 43, Jun 85, Jul 68, Aug 60, Sep 15 (to 2026-09-12). None between 2026-09-12 and 2026-10-10. In 2025 there were 3,008. Four of 24 policy servers were online in the 30 days to 2026-10-11.",
     "short": "Active. The site says it will run through December 2026."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Evaluators write a natural-language instruction for each A/B test; tasks are table and room manipulation (pick and place, open and close, wiping, folding, tool use)."
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "object-instance",
      "visual",
      "language",
      "new-task"
     ],
     "display": "Evaluators pick any scene, objects, camera view and instruction. Variation is wide but uncontrolled and not labelled.",
     "level": "inferred",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Paper Figure 11: lighting, camera viewpoints, tablecloths and objects were often changed. The 3,995 public evaluations use 2,797 distinct instruction strings (our count). RoboArena does not record whether a task or object resembles the policy's training data.",
     "short": "Open-ended. Evaluators choose the scenes and tasks."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Closed-loop runs on physical DROID robots."
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "One Franka arm on the DROID platform. The authors list evaluation on DROID only as a limitation."
    },
    "robots": {
     "value": "Franka Panda (DROID setup)",
     "display": "DROID platform: Franka Panda 7-DoF arm, Robotiq 2F-85 gripper, ZED-mini wrist camera, one or more ZED 2 external cameras, on a height-adjustable mobile table.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Evaluators use their own labs, offices and kitchens. The paper reports dozens of scenes and shows 32 sample environments (Figure 11)."
    },
    "tasks": {
     "value": "open-ended",
     "display": "No fixed task list. The paper reports hundreds of instructions; the public evaluations use 2,797 distinct instruction strings.",
     "level": "inferred",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Distinct strings counted by us on 2026-10-10; after lower-casing and removing punctuation, 2,761. Most frequent: 'open the book' (40), 'close the book' (35).",
     "short": "Open-ended, with 2,797 distinct instructions"
    },
    "demonstrations": {
     "value": "none provided",
     "display": "RoboArena supplies no training data. It points submitters to the open DROID dataset and the openpi DROID training code.",
     "level": "verified",
     "sources": [
      "s5",
      "s19"
     ],
     "checked": "2026-10-10",
     "short": "None. Submitters use DROID data."
    },
    "scale": {
     "value": "3,994 counted A/B evaluations, 24 policies",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper (2025)",
       "display": "7 policies, 612 A/B comparisons, 4,284 evaluation rollouts including the oracle runs, 7 universities.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Live service (2026-10-10)",
       "display": "3,994 counted A/B evaluations (575 ties, 14.4%), 24 policies, 120 policy pairs, 60 evaluator accounts, 28 evaluator-organisation labels. First evaluation 2025-04-15, last 2026-09-12.",
       "level": "verified",
       "sources": [
        "s8"
       ],
       "note": "The 28 labels include 11 spellings of Berkeley and two non-institution labels ('CS224R Project', 'CoRL 2025 Seoul'). Merging the Berkeley spellings gives 18 labels (our count)."
      },
      {
       "value": "Data dump 2026-07-17",
       "display": "3,883 evaluation sessions, 10,783 policy episodes, 27,148 videos, 21.7 GB (21,676,174,280 bytes).",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Removed in June 2026",
       "display": "847 evaluations dated 2026-04-02 to 2026-06-03 were taken out of the public list and the ranking (issues.i2).",
       "level": "inferred",
       "sources": [
        "s13",
        "s9"
       ]
      }
     ],
     "short": "3,994 A/B evaluations of 24 policies"
    },
    "scoring": {
     "value": [
      "preference",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "For each A/B test the evaluator gives a preference (A, B or tie in the live data), a 0 to 100 progress score per policy, and a written reason. The paper found preference and progress disagree in 11% of A/B evaluations. The live rating is described as a Bradley-Terry-Davidson fit to preferences; we did not find whether progress scores enter it."
    },
    "metric_detail": {
     "value": "pairwise rating with standard deviation",
     "display": "Each policy gets a rating on an Elo-like scale (786 to 1788 on 2026-10-10) with a standard deviation, fitted to all counted A/B preferences, ties included. The official view lists only policies with at least 100 A/B evaluations.",
     "level": "verified",
     "sources": [
      "s5",
      "s6",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper's preferred model adds latent task-difficulty buckets fitted by an EM algorithm (Section 3.2, Appendix B). The site calls the live method a 'Bradley-Terry Davidson ranking path' and labels its chart 'Elo score'. Ratings are relative: other policies' results move every value (issues.i5). Comparisons involving the base pi0 or pi0-FAST models are shown but not counted.",
     "short": "A rating from pairwise comparisons, with a standard deviation. The official view lists only policies with at least 100 tests."
    },
    "trials": {
     "value": "one rollout per policy per A/B test",
     "display": "Each A/B evaluation runs each of the two policies once, back to back, from the same start. The official view needs at least 100 A/B evaluations per policy. In the paper's oracle sessions, evaluators ran all 7 policies on each task.",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "Each policy runs once in each A/B test"
    },
    "uncertainty_reported": {
     "value": "yes",
     "level": "verified",
     "sources": [
      "s6",
      "s4",
      "s33"
     ],
     "checked": "2026-10-10",
     "note": "The leaderboard shows a standard deviation for every rating, and the paper's figures carry error bars. PhAIL (2026-05) says neither RoboArena nor RoboChallenge reports confidence intervals or paired tests on the rankings."
    },
    "evaluator": {
     "value": "organiser-run",
     "display": "Volunteer evaluators run the tests on their own DROID robots; submitters host their own policy servers; RoboArena's central server assigns pairs and computes the ranking.",
     "level": "inferred",
     "sources": [
      "s2",
      "s5",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Step by step: (1) A submitter hosts the policy on a remote server and submits it through a form; the team checks the input and output format and runs it with a trained evaluator in a test setup before adding it to the pool. (2) An evaluator's client asks the central server for a pair and receives the two servers' IP addresses, not the policy names. (3) The evaluator arranges a scene, types an instruction and runs policy A and then policy B from the same start until success or a timeout. (4) The evaluator records a preference, two progress scores and a reason; videos and actions are uploaded. (5) The central server fits the rating. Evaluators earn evaluation credits; submitters get a weekly budget. Since June 2026 only evaluators with no stake in submitted policies may volunteer. Taxonomy fit is loose: the organisers coordinate but do not run the robots."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s6",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Public leaderboard at robo-arena.github.io/leaderboard, with an A/B evaluation viewer and per-organisation audit metrics."
    },
    "top_score": {
     "value": 1735,
     "display": "DreamZero (NVIDIA) 1735 ± 42.6 from 190 A/B evaluations, first in the official view (read 2026-10-10). In the all-policies view, Spirit v1.6 leads with 1788 ± 106 from 25 evaluations.",
     "level": "verified",
     "sources": [
      "s6",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Read from the leaderboard data at 2026-10-11 03:30 UTC. Display names come from the site's alias table (backend names dreaming_zebra and j2-vla). Ratings are relative and have no ceiling, so these items carry no chart data.",
     "items": [
      {
       "value": "DreamZero: 1735 ± 42.6 (official #1)",
       "display": "190 A/B evaluations, all between 2026-01-28 and 2026-03-30. Its policy server was offline throughout the 30 days to 2026-10-11.",
       "level": "verified",
       "sources": [
        "s6",
        "s9",
        "s23"
       ]
      },
      {
       "value": "pi0.5-DROID: 1608 ± 30.7 (official #2)",
       "display": "745 A/B evaluations; Physical Intelligence policy; server online.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "pi0-FAST-DROID: 1582 ± 29.6 (official #3)",
       "display": "941 A/B evaluations. It ranked first in the paper's 2025 oracle ranking.",
       "level": "verified",
       "sources": [
        "s6",
        "s2"
       ]
      },
      {
       "value": "Spirit v1.6: 1788 ± 106 (all-policies #1)",
       "display": "25 counted evaluations, 24 of them from one evaluator label, 2026-04-30 to 2026-05-15. Below the 100-evaluation threshold. Developer Spirit AI per press reports.",
       "level": "verified",
       "sources": [
        "s6",
        "s9",
        "s30"
       ]
      },
      {
       "value": "Apricot: 1661 ± 61.0 (all-policies #3)",
       "display": "63 A/B evaluations; marked closed-source.",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "Before the rollback: Spirit v1.6 1926 ± 37.3",
       "display": "First place with 310 A/B evaluations in the leaderboard data archived on 2026-06-03. 284 of its evaluations were later removed (issues.i2).",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "note": "Read from an Internet Archive copy of the official endpoint."
      }
     ],
     "short": "DreamZero, rated 1735 in the official view (October 2026)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s18",
      "s19",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: 'Copyright (c) 2025 robo-arena'. It covers the public repository, which holds the policy-server interface, client example and a test script. The evaluator client, central server and ranking code are not in it (file tree checked 2026-10-10). The website repository and the linked DROID simulation repository (arhanjain/sim-evals) have no licence file."
    },
    "license_data": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s20",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "All three Hugging Face data dumps carry license: mit. The paper text is CC BY 4.0 on arXiv."
    },
    "license_assets": {
     "value": "not applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Tests use physical robots and objects; no 3D assets are distributed."
    },
    "access": {
     "value": "application",
     "display": "Policies are submitted through an online form and checked before they enter the pool; evaluators also sign up through a form. Results, videos and data dumps are open to all.",
     "level": "verified",
     "sources": [
      "s5",
      "s14",
      "s21"
     ],
     "checked": "2026-10-10",
     "short": "Policies are submitted through a form. The data is open."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s18",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Code and data dumps are MIT-licensed. Not legal advice. The dumps contain videos recorded in evaluators' labs."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s2",
      "s27",
      "s28",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from real robots. RoboArena links a DROID simulation harness (sim-evals) for debugging only. Other groups use RoboArena as the real-world reference for their own tools: PolaRiS reports Pearson r 0.98 against RoboArena progress scores; RoboLab-120 reports Spearman 1.00 and Pearson 0.68 against RoboArena ratings for 4 policies; RoboWorld reports Pearson 0.989 and Spearman 0.970 against the 2026-02-26 leaderboard for 8 policies. These test the other tools, not RoboArena. For checks of RoboArena's own ranking, see validity."
    },
    "real_reproducibility": {
     "value": "multi-site-measured",
     "display": "The same policies are tested at many sites, and the site publishes leave-one-organisation reruns of the official ranking.",
     "level": "inferred",
     "sources": [
      "s2",
      "s8",
      "s10",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Paper: evaluators at 7 universities, compared against an exhaustive oracle (validity). Live: 28 organisation labels, but one label supplies 57.6% of counted evaluations (issues.i3). No published table compares the same policies' scores site by site. Within one A/B test, the evaluator matches the start state for both policies; across tests, conditions are not matched by design."
    },
    "published_at": {
     "value": "CoRL 2025 (PMLR 305:336-364)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "arXiv comments give no venue; the PMLR page confirms."
    },
    "citations": {
     "value": 87,
     "display": "87 (Semantic Scholar; 11 influential)",
     "level": "verified",
     "sources": [
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "Read 2026-10-11 03:56 UTC.",
     "short": "87"
    },
    "github_stars": {
     "value": 116,
     "display": "116 stars, 10 forks (robo-arena/roboarena)",
     "level": "verified",
     "sources": [
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "116"
    },
    "dataset_downloads": {
     "value": 9021,
     "display": "9,021 (Hub 'downloads' field), 28,474 all time: DataDump_07-17-2026",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "We did not check the time window behind the 'downloads' field. The two earlier dumps show 997 and 1,123.",
     "short": "9,021 on Hugging Face"
    },
    "used_by": {
     "value": "Real-world reference for at least 3 evaluation papers; cited by Physical Intelligence and NVIDIA for their DROID policies.",
     "level": "verified",
     "sources": [
      "s27",
      "s28",
      "s29",
      "s26",
      "s25"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "PolaRiS (2025-12)",
       "display": "Real-to-sim evaluation; Pearson r 0.98 against RoboArena average progress scores.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "RoboLab-120 (NVIDIA, 2026-04)",
       "display": "Simulation benchmark; Spearman 1.00 and Pearson 0.68 against RoboArena ratings for 4 policies.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "RoboWorld (KAIST, Config, 2026-07)",
       "display": "World-model evaluator; Pearson 0.989 and Spearman 0.970 against the 2026-02-26 leaderboard for 8 policies, using the 2026-02-03 data dump.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "openpi (Physical Intelligence)",
       "display": "Calls pi0.5-DROID its strongest generalist DROID policy 'based on the public RoboArena benchmark' and hosts the paper's baseline checkpoints.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "DreamZero (NVIDIA, 2026-02)",
       "display": "Integrated into RoboArena (paper acknowledgments); a project lead posted that it was first on RoboArena (2026-02-27); an NVIDIA blog cites 1750 Elo on the April 2026 leaderboard against 1622 for pi0.5.",
       "level": "verified",
       "sources": [
        "s23",
        "s24",
        "s25"
       ]
      },
      {
       "value": "CoRL 2025 RoboArena challenge",
       "display": "Policy development challenge; final RoboArena evaluations ran 2025-09-13 to 2025-09-20.",
       "level": "verified",
       "sources": [
        "s32",
        "s5"
       ]
      }
     ],
     "short": "Real-world reference for at least 3 evaluation papers"
    },
    "industry_use": {
     "value": [
      "Physical Intelligence",
      "NVIDIA",
      "Spirit AI",
      "FrodoBots"
     ],
     "level": "verified",
     "sources": [
      "s26",
      "s23",
      "s25",
      "s30",
      "s8"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Physical Intelligence",
       "display": "pi0, pi0-FAST and pi0.5 DROID policies are in the pool; openpi points to RoboArena.",
       "level": "verified",
       "sources": [
        "s26",
        "s6"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "DreamZero leads the official view; NVIDIA-labelled evaluators ran 219 counted evaluations in January and February 2026.",
       "level": "verified",
       "sources": [
        "s23",
        "s8"
       ]
      },
      {
       "value": "Spirit AI",
       "display": "Announced first place for Spirit v1.6 on 2026-06-03 (press reports); most of its evaluations were later removed.",
       "level": "reported",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "FrodoBots",
       "display": "Evaluator-organisation label with 2,301 of 3,994 counted evaluations (self-reported label in public data).",
       "level": "verified",
       "sources": [
        "s8"
       ]
      }
     ]
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Evaluations were manipulated in spring 2026",
     "text": "On 2026-06-13 the RoboArena team said it had observed evidence of benchmark manipulation since April. Its notice says one sign was unusually low completion rates for requested evaluation assignments, and that organisations completing less than 20% of their requested evaluations were flagged. The team excluded evaluations from organisations with suspicious patterns and now allows only third-party evaluators with no stake in submitted policies. Both changes were applied retroactively to evaluations from 2026-04-02; the team says it found no evidence requiring exclusion before that date. It also added a 100-evaluation threshold for the official view, audit metrics and easier data downloads. The notice was removed from the site on 2026-07-17.",
     "level": "verified",
     "sources": [
      "s14",
      "s16",
      "s15",
      "s31"
     ],
     "status": "addressed",
     "mitigation": {
      "text": "Audit metrics on the site show each organisation's share of evaluations, single-organisation dependence per policy and leave-one-organisation reruns of the official ranking.",
      "sources": [
       "s5",
       "s10"
      ]
     },
     "note": "Notice text read from the website source at commit a6602e6 (2026-06-13). Detection relies on a completion-rate threshold and the no-stake rule; the notice names no organisations.",
     "short": "The organisers found manipulation, rolled back evaluations from April 2026 and changed the rules."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "The rollback removed 847 evaluations and changed the top of the board",
     "text": "Comparing the official data archived on 2026-06-03 with today's data: 847 of the 4,614 public evaluations then listed are gone, all dated 2026-04-02 to 2026-06-03. They came from six evaluator-organisation labels (333, 276, 142, 72, 22 and 2 evaluations); of the 954 evaluations listed for that period, 107 remain. Spirit v1.6 led with 1926 ± 37.3 from 310 evaluations; 223 of its 309 listed evaluations came from two labels that first appeared in April and May 2026 and recorded 212 wins, 1 loss and 10 ties for it, while all other evaluators recorded 33 wins, 37 losses and 16 ties. After the rollback it has 25 counted evaluations and appears only in the all-policies view. WALL-OSS (fourth on 2026-06-03, 48 evaluations) no longer appears. MolmoAct2-DROID went from 63 evaluations and 1066 to 5 evaluations and 1579. Press reports say Spirit AI had announced first place on 2026-06-03.",
     "level": "inferred",
     "sources": [
      "s12",
      "s13",
      "s9",
      "s6",
      "s30"
     ],
     "status": "addressed",
     "note": "Our comparison by session ID of two copies of the official evaluation list. The archived copies were made by the Internet Archive. The earliest evaluation of one of the two labels is dated 2026-04-02, the rollback cut-off date. We do not name the labels because the organisers' notice does not say which rule removed which evaluations.",
     "short": "847 evaluations from April to June 2026 were removed. Spirit v1.6, which led the board before, now has 25 counted tests."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Most evaluations come from one evaluator label",
     "text": "One evaluator-organisation label, 'frodobots', contributed 2,301 of 3,994 counted evaluations (57.6%) between 2025-08-02 and 2026-09-12. Its share was 659 of 987 in 2026 (66.8%) and 195 of 210 after 2026-06-13 (92.9%). For every official policy except DreamZero, frodobots supplied the largest share of its evaluations (53.9% to 67.5%). The official leave-one-organisation rerun without this label keeps the top two policies but moves others by up to two places.",
     "level": "inferred",
     "sources": [
      "s8",
      "s9",
      "s10"
     ],
     "status": "open",
     "note": "Shares computed by us from the public statistics; the rerun is the site's own. The label is self-reported. We found no primary source describing this organisation's role.",
     "short": "One evaluator label supplied 57.6% of all counted evaluations and 92.9% since mid-June 2026."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The official leader was evaluated mostly under its developer's label",
     "text": "DreamZero, an NVIDIA model, has 190 counted evaluations; 111 of them (58.4%) carry the evaluator label 'nvidia' and date from 2026-01-28 to 2026-02-26. These predate the 2026-04-02 start of the retroactive no-stake rule and remain counted. The site's own leave-one-organisation rerun without 'nvidia' leaves DreamZero with 79 evaluations, below the 100-evaluation threshold, so pi0.5-DROID would lead the official view. DreamZero has had no new evaluations since 2026-03-30 and its server was offline in the 30 days to 2026-10-11.",
     "level": "inferred",
     "sources": [
      "s9",
      "s23",
      "s11",
      "s14",
      "s6"
     ],
     "status": "open",
     "note": "Evaluations are double-blind by design: evaluators receive server addresses, not policy names. We make no claim about how any evaluation was done.",
     "short": "58% of the official leader's evaluations carry its developer's organisation label."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Ratings shift with the pool, and one weekly snapshot was broken",
     "text": "Ratings are relative. In the published weekly snapshots, policies with no new evaluations still change score; for example Spirit v1.6 ranged from 1773 to 1792 between June and October 2026. On the 2026-08-24 snapshot every policy rose by 244 to 278 points and fell back by about 233 a week later; that snapshot gave a newly added policy with 2 evaluations a score of -3884 and a standard deviation of 39,859.4, and listed a standard deviation of about 1733 for every other policy.",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "status": "open",
     "note": "Read by us from the official snapshot history (18 weekly snapshots, 2026-06-10 to 2026-10-05). The current board looks normal.",
     "short": "Scores move when other policies are added, and one weekly snapshot shifted every score by about 240 points."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Limitations stated by the authors",
     "text": "The paper lists: evaluation on the DROID platform only; difficulty running controlled experiments that vary one factor at a time; robustness to intentionally adversarial evaluators not studied; possible over-optimisation (Goodhart's law); and possible latency from remote inference, found negligible for static tasks. It also reports that preference and progress feedback disagreed in 11% of A/B evaluations.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "The authors say RoboArena covers only the DROID platform and cannot easily run controlled tests that change one factor at a time. They did not study how robust it is to evaluators who act in bad faith."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A RoboArena rating says how often evaluators preferred a policy over the others in the pool, on DROID Franka arms, in tasks the evaluators chose. It is a relative number that moves when the pool or the evaluators change, and it does not translate into a success rate or carry over to other robots.",
     "basis": [
      "facts.metric_detail",
      "facts.embodiment",
      "issues.i5"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A rating shows how often evaluators preferred a policy in this pool, on this robot."
    },
    {
     "id": "r2",
     "text": "The top of the board rests on thin evidence. The official leader's place depends on evaluations carrying its developer's label, and the all-policies leader has 25 evaluations from essentially one label. Read the evaluation counts and the audit metrics next to every rating.",
     "basis": [
      "facts.top_score",
      "issues.i3",
      "issues.i4"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check evaluation counts and evaluator shares before trusting the top places."
    },
    {
     "id": "r3",
     "text": "The paper's accuracy check compares RoboArena rankings with an oracle ranking (a reference built from the same evaluators, tasks and progress scores), over 7 related policies. It shows that the pairwise method recovers that oracle efficiently. It is not a comparison with an independent measure of real-world usefulness.",
     "basis": [
      "facts.scoring",
      "sources.s2",
      "sources.s4"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The accuracy check uses only RoboArena's own data."
    },
    {
     "id": "r4",
     "text": "Public evaluation data let the organisers and outsiders spot the spring 2026 manipulation, and the rollback can be traced in archived data. The fixes are rules and thresholds added afterwards. A large share of evaluations still comes from a few evaluators, so the board stays vulnerable to them.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Open data made the manipulation visible. A few evaluators still supply a large share of the tests, which remains a risk."
    }
   ],
   "searched": [
    {
     "for": "independent checks of the ranking method",
     "where": "Semantic Scholar list of all 87 citing papers (read 2026-10-10) and full texts of PolaRiS, RoboLab, RoboWorld, PhAIL and VLA-REPLICA; web searches for critiques of RoboArena's pairwise method. Found uses of RoboArena as a reference and PhAIL's remark on confidence intervals, but no independent re-test of the ranking against another real-world measure.",
     "date": "2026-10-10"
    },
    {
     "for": "live ranking model",
     "where": "Site code and text (calls it 'Bradley-Terry Davidson'), paper Section 3.2 and Appendix B, public repository (no ranking code). Whether the live fit uses the paper's task buckets or progress scores is not stated.",
     "date": "2026-10-10"
    },
    {
     "for": "which organisations were excluded and why",
     "where": "Integrity notice at website commits bda3698 and a6602e6, the lead author's post, current site. The organisers name no organisations; press reports name some (secondary).",
     "date": "2026-10-10"
    },
    {
     "for": "role of the 'frodobots' evaluator label",
     "where": "Paper, site text, transparency data, web search. No primary source describes it.",
     "date": "2026-10-10"
    },
    {
     "for": "number of scenes in the live service",
     "where": "Paper (dozens; 32 shown), data dump card, transparency data. Not counted anywhere.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RoboArena: Distributed Real-World Evaluation of Generalist Robot Policies (arXiv abstract page, v1 2025-06-22, v2 2025-11-29)",
     "url": "https://arxiv.org/abs/2506.18123",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RoboArena paper, full text v2 (Sections 3-7, Appendices B-D)",
     "url": "https://arxiv.org/html/2506.18123v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "RoboArena, Proceedings of the 9th Conference on Robot Learning, PMLR 305:336-364",
     "url": "https://proceedings.mlr.press/v305/atreya25a.html",
     "type": "paper",
     "publisher": "PMLR (CoRL 2025)",
     "date": "2025-09",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "RoboArena paper Figure 6: ranking accuracy versus the oracle (Pearson r and MMRV bar labels; identical in v1 and v2)",
     "url": "https://arxiv.org/html/2506.18123v2/roboarena_ranking_results.svg",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RoboArena website and its page code (about text, leaderboard views, alias table, audit metrics)",
     "url": "https://robo-arena.github.io/assets/index-FWyWX4NQ.js",
     "type": "site",
     "publisher": "RoboArena team",
     "date": "2026-07-17",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RoboArena leaderboard data (JSON loaded by robo-arena.github.io/leaderboard; last_updated 2026-10-11 03:30 UTC)",
     "url": "https://roboarena-api-domain-name.online/api/leaderboard",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "RoboArena weekly leaderboard snapshots (18 snapshots, 2026-06-10 to 2026-10-05)",
     "url": "https://roboarena-api-domain-name.online/api/leaderboard_history",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "RoboArena public transparency statistics (counted A/B evaluations by policy, pair and evaluator organisation)",
     "url": "https://roboarena-api-domain-name.online/api/transparency_summary",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "RoboArena public A/B evaluation list (3,995 sessions with instruction, preference, progress scores and evaluator organisation label)",
     "url": "https://roboarena-api-domain-name.online/api/list_ab_evaluations",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "RoboArena official leave-one-organisation rerun: label 'frodobots' removed",
     "url": "https://roboarena-api-domain-name.online/api/evaluator_leaderboard_impact?org=frodobots",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "RoboArena official leave-one-organisation rerun: label 'nvidia' removed",
     "url": "https://roboarena-api-domain-name.online/api/evaluator_leaderboard_impact?org=nvidia",
     "type": "leaderboard",
     "publisher": "RoboArena team",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "RoboArena leaderboard data as archived on 2026-06-03 (last_updated 2026-06-03 11:49 UTC), before the rollback",
     "url": "https://web.archive.org/web/20260603123403/https://roboarena-api-domain-name.online/api/leaderboard",
     "type": "leaderboard",
     "publisher": "RoboArena team (copy held by the Internet Archive)",
     "date": "2026-06-03",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "RoboArena public A/B evaluation list as archived on 2026-06-03 (4,614 sessions), before the rollback",
     "url": "https://web.archive.org/web/20260603124400/https://roboarena-api-domain-name.online/api/list_ab_evaluations",
     "type": "leaderboard",
     "publisher": "RoboArena team (copy held by the Internet Archive)",
     "date": "2026-06-03",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "RoboArena benchmark integrity notice, website source at commit a6602e6 ('Publish post-maintenance site')",
     "url": "https://github.com/robo-arena/robo-arena.github.io/blob/a6602e6c6785cd1f3a79596df2fa26073e2bce07/src/components/BenchmarkIntegrityNotice.jsx",
     "type": "site",
     "publisher": "RoboArena team",
     "date": "2026-06-13",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "RoboArena website repository commit history (notice added 2026-06-11; 100-evaluation official view 2026-06-10; notice hidden 2026-07-17)",
     "url": "https://github.com/robo-arena/robo-arena.github.io/commits/main",
     "type": "repo",
     "publisher": "RoboArena team",
     "date": "2026-07-17",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Post by Pranav Atreya (RoboArena lead author) on benchmark hacking and rolled-back evaluations",
     "url": "https://x.com/pranav_atreya/status/2065605060084871170",
     "type": "blog",
     "publisher": "Pranav Atreya (UC Berkeley)",
     "date": "2026-06-13",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "GitHub API: robo-arena/roboarena (stars, forks, created, pushed; no releases or tags)",
     "url": "https://api.github.com/repos/robo-arena/roboarena",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "robo-arena/roboarena LICENSE file",
     "url": "https://github.com/robo-arena/roboarena/blob/main/LICENSE",
     "type": "repo",
     "publisher": "RoboArena team",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "robo-arena/roboarena README and file tree (policy server interface only)",
     "url": "https://github.com/robo-arena/roboarena",
     "type": "repo",
     "publisher": "RoboArena team",
     "date": "2026-04-28",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Hugging Face API: datasets by RoboArena (three data dumps, licence tags)",
     "url": "https://huggingface.co/api/datasets?author=RoboArena&full=true",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-07-17",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "RoboArena/DataDump_07-17-2026 dataset card and Hub record (size, downloads)",
     "url": "https://huggingface.co/datasets/RoboArena/DataDump_07-17-2026",
     "type": "repo",
     "publisher": "RoboArena team",
     "date": "2026-07-17",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Semantic Scholar API record for arXiv:2506.18123",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2506.18123?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "World Action Models are Zero-shot Policies (DreamZero), arXiv 2602.15922 v1",
     "url": "https://arxiv.org/html/2602.15922v1",
     "type": "paper",
     "publisher": "NVIDIA",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Post by Joel Jang (DreamZero project lead): 'DreamZero is #1 on both MolmoSpaces and RoboArena'",
     "url": "https://x.com/jang_yoel/status/2027525664497397861",
     "type": "blog",
     "publisher": "Joel Jang (NVIDIA)",
     "date": "2026-02-27",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Pretrained to Imagine, Fine-Tuned to Act: The Rise of World-Action Models (NVIDIA Technical Blog)",
     "url": "https://developer.nvidia.com/blog/pretrained-to-imagine-fine-tuned-to-act-the-rise-of-world-action-models/",
     "type": "blog",
     "publisher": "NVIDIA",
     "date": "2026-06-15",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "openpi DROID example README (RoboArena baseline checkpoints; pi0.5-DROID named strongest based on RoboArena)",
     "url": "https://github.com/Physical-Intelligence/openpi/blob/main/examples/droid/README.md",
     "type": "repo",
     "publisher": "Physical Intelligence",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (v2; Figure 8, correlation with RoboArena)",
     "url": "https://arxiv.org/html/2512.16881v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies (v4; Fig. 10, RoboArena Elo comparison)",
     "url": "https://arxiv.org/html/2604.09860v4",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation (v4; RoboArena leaderboard as ground truth)",
     "url": "https://arxiv.org/html/2607.01060v4",
     "type": "paper",
     "publisher": "arXiv (KAIST, Config)",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Huxiu: report on RoboArena manipulation and Spirit v1.6 (republished from WeChat account Dianchang)",
     "url": "https://www.huxiu.com/article/4869933.html",
     "type": "secondary",
     "publisher": "Huxiu",
     "date": "2026-06-24",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "What Do Robotics Leaderboards Tell Us About The State of Robot Learning? (It Can Think! newsletter)",
     "url": "https://itcanthink.substack.com/p/what-do-robotics-leaderboards-tell",
     "type": "secondary",
     "publisher": "Chris Paxton",
     "date": "2026-06-13",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "RoboArena Workshop and CoRL 2025 challenge page",
     "url": "https://sites.google.com/view/corl-roboarena",
     "type": "site",
     "publisher": "RoboArena organisers",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "PhAIL: A Real-Robot VLA Benchmark and Distributional Methodology (Related Work)",
     "url": "https://arxiv.org/html/2605.29710",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "GitHub API: arhanjain/sim-evals (DROID simulated evaluation linked by RoboArena; no licence)",
     "url": "https://api.github.com/repos/arhanjain/sim-evals",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the checked basic entry. Added the June 2026 manipulation notice and rollback (site source history, lead author's post, archived official data), evaluator-concentration and self-evaluation issues from the public statistics, the leave-one-organisation reruns, the snapshot-scale anomaly, Figure 6 values, and adoption sources. The basic entry's 'official view' figures were re-read and are unchanged."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How a policy performs on other robots.",
     "sub": "RoboArena tests only the DROID Franka setup.",
     "basis": [
      "facts.embodiment",
      "issues.i6"
     ]
    },
    {
     "id": "l2",
     "text": "How often a policy succeeds on a fixed set of tasks.",
     "sub": "Scores are relative ratings from tests where evaluators choose the tasks.",
     "basis": [
      "facts.metric_detail",
      "facts.tasks",
      "issues.i5"
     ]
    },
    {
     "id": "l3",
     "text": "How good the rarely tested policies are.",
     "sub": "Of the 24 policies, 15 have fewer than 100 evaluations.",
     "basis": [
      "facts.top_score",
      "facts.scale"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "Oracle ranking check (paper Figure 6)",
     "date": "2025-06",
     "by": "authors",
     "method": "Rankings built from 612 A/B tests of 7 DROID policies were compared with an 'oracle' ranking. After each A/B test, the evaluator also ran the other 5 policies on the same task. The oracle ranks policies by their average progress score over 4,284 rollouts.",
     "result": "Pearson r = 0.98 and MMRV = 1.8% for the task-aware model. A conventional 17-task lab protocol gave r = 0.69 and MMRV = 13%. MMRV measures how often two rankings disagree.",
     "authors_view": "more accurately rank",
     "n_policies": 7,
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "note": "Values read from the figure's bar labels. Elo r 0.90 (MMRV 5.7%), plain Bradley-Terry r 0.97 (2.7%), progress-score averaging r 0.98 (1.1%). The oracle uses the same evaluators, tasks and progress scores as RoboArena, so this checks internal consistency over RoboArena's own task mix. The conventional baseline (17 tasks, 44 episodes per policy) is scored against that same oracle. Figure 7: ranking quality converges within about 100 A/B tests."
    },
    {
     "id": "v2",
     "name": "Simulated drift re-analysis (paper Table 2, arXiv v2)",
     "date": "2025-11",
     "by": "authors",
     "method": "The same data were ranked again after simulating a shift over time from easy to hard tasks and from weaker to stronger policies. The new rankings were compared with the oracle.",
     "result": "Pearson r = 0.838 and MMRV = 0.058. The conventional protocol gave r = 0.692 and MMRV = 0.141.",
     "authors_view": "significantly higher correlation",
     "n_policies": 7,
     "level": "verified",
     "sources": [
      "s2"
     ]
    },
    {
     "id": "v3",
     "name": "Leave-one-organisation rerun (live audit tool)",
     "date": "2026-06",
     "by": "authors",
     "method": "The organisers' site recomputes the official ranking with all evaluations from one evaluator organisation removed. We read the reruns for the two most influential labels on 2026-10-10.",
     "result": "Without the largest label (57.6% of evaluations), ranks move by at most 2 places. Without 'nvidia', DreamZero falls below 100 evaluations.",
     "n_policies": 9,
     "level": "verified",
     "sources": [
      "s10",
      "s11",
      "s15"
     ],
     "note": "Without 'frodobots', paligemma_fast_specialist_droid moves from 6th to 4th and pi0-FAST-DROID from 3rd to 5th; ratings shift by up to 468 points because the scale is relative. Without 'nvidia', pi0.5-DROID would lead. A rerun without 'Berkeley' changes no rank."
    }
   ]
  },
  {
   "id": "robocasa",
   "name": "RoboCasa",
   "full_name": "RoboCasa: Large-Scale Simulation of Everyday Tasks for Generalist Robots",
   "aliases": [
    "RoboCasa (original, 2024)",
    "RoboCasa v0.1/v0.2",
    "RoboCasa RSS24 protocol",
    "RoboCasa Kitchen (24 tasks)",
    "RoboCasa: Large-Scale Simulation of Household Tasks for Generalist Robots"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Simulation framework with 100 tasks defined for systematic evaluation of robot policies by success rate; its 24 atomic tasks are a widely reported benchmark.",
   "summary": {
    "text": "RoboCasa is a kitchen simulator from UT Austin and NVIDIA with 100 tasks, 120 kitchens and 2,509 objects for a mobile robot arm. Most papers report the average success rate on its 24 short atomic tasks; the 2026 successor RoboCasa365 replaces it.",
    "short": "RoboCasa is a kitchen simulator with 100 tasks for a mobile robot arm. Most papers use it to report the average success rate on 24 short tasks.",
    "sources": [
     "s2",
     "s6",
     "s21"
    ]
   },
   "facts": {
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper calls RoboCasa a simulation framework and defines 100 tasks 'for systematic evaluation'. Later papers use its 24 atomic tasks as a benchmark.",
     "short": "Kitchen simulator with fixed tasks"
    },
    "kind_secondary": {
     "value": [
      "benchmark",
      "dataset"
     ],
     "display": "Also a task benchmark (24 atomic tasks in common use) and a demonstration dataset (human and MimicGen-generated)",
     "level": "verified",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "v0.2",
     "display": "v0.2 (2024-10-31): robosuite v1.5 backend. Tagged on 2025-12-18 and published as a GitHub release on 2026-02-17. The 2024 paper used v0.1, which has no tag.",
     "level": "verified",
     "sources": [
      "s5",
      "s8",
      "s9",
      "s10"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v0.1 (2024-06)",
       "display": "No tag. The maintainer points to commit 0f604b25 (2024-10-28) and the robocasa_v0.1 branch of robosuite.",
       "level": "verified",
       "sources": [
        "s10"
       ]
      },
      {
       "value": "v0.2 branch changes",
       "display": "Bug fix on 2025-02-27 changes how object rotations are sampled at reset (quaternion order in the placement sampler); default renderer changed 2025-03-18.",
       "level": "verified",
       "sources": [
        "s9"
       ]
      },
      {
       "value": "NVIDIA fork",
       "display": "Isaac-GR00T evaluates 'RoboCasa Kitchen' with squarefk/robocasa, 4 commits ahead of v0.2 (gymnasium wrapper; reward returns 1 on success), last commit 2025-11-06.",
       "level": "verified",
       "sources": [
        "s34",
        "s35"
       ]
      },
      {
       "value": "RoboCasa365 v1.0 (2026-02-18)",
       "display": "Successor on the same repository's main branch; v1.0.1 (2026-05-12) raised task horizons 1.5x. See facts.successor.",
       "level": "verified",
       "sources": [
        "s6",
        "s8"
       ]
      }
     ],
     "short": "v0.2 (October 2024)"
    },
    "successor": {
     "value": "RoboCasa365",
     "display": "RoboCasa365 (arXiv 2026-03, ICLR 2026) is built on the RoboCasa framework by the same lab: 365 tasks, 2,500 pretraining kitchens plus 10 target kitchens reusing the original layouts and styles, the original 2,509 objects plus new ones, and an official leaderboard on 50 tasks. It is a separate Atlas entry.",
     "level": "verified",
     "sources": [
      "s16",
      "s6",
      "s17",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Task names such as PickPlaceCounterToCabinet persist in RoboCasa365 code with new scenes, splits and horizons, so a result run on current code is not the 2024 protocol. RoboCasa365's paper describes the original as '100k demonstrations spanning 30 tasks and 100 scenes'; the original paper says 120 scenes.",
     "short": "Replaced by RoboCasa365 in 2026"
    },
    "publishers": {
     "value": [
      "The University of Texas at Austin",
      "NVIDIA Research"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Soroush Nasiriany, Abhiram Maddukuri, Lance Zhang, Adeet Parikh, Aaron Lo, Abhishek Joshi, Ajay Mandlekar, Yuke Zhu.",
     "items": [
      {
       "value": "The University of Texas at Austin",
       "display": "7 of 8 authors (Robot Perception and Learning Lab)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "NVIDIA Research",
       "display": "Ajay Mandlekar (NVIDIA only) and Yuke Zhu (UT Austin and NVIDIA)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "University lab lead with NVIDIA co-authors."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-06",
     "display": "arXiv v1 2024-06-04 (only version); repository created 2024-05-11; RSS 2024.",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "The RSS proceedings title says 'Household Tasks' where arXiv says 'Everyday Tasks'.",
     "short": "June 2024, at RSS 2024"
    },
    "latest_update": {
     "value": "2025-03",
     "display": "Last code change on the v0.2 branch 2025-03-18 (default renderer); docs fix 2025-04-23; README edit 2025-12-18. Main moved to RoboCasa365 on 2026-02-18.",
     "level": "verified",
     "sources": [
      "s9",
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "March 2025. The repository then moved on to RoboCasa365."
    },
    "status": {
     "value": "superseded",
     "display": "Superseded by RoboCasa365 in the same repository. Its 24-task protocol is still widely reported.",
     "level": "inferred",
     "sources": [
      "s6",
      "s8",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "The main README calls RoboCasa365 'the latest iteration' of RoboCasa. The audit tracker counts 12 papers in March 2026 and 11 in April 2026 reporting RoboCasa results (all protocols).",
     "short": "Replaced by RoboCasa365. Its 24-task protocol is still widely used."
    },
    "capability": {
     "value": [
      "manipulation",
      "mobile-manipulation",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Eight skills: pick and place, doors, drawers, knobs, levers, buttons, insertion, navigation. Composite tasks chain skills; policies are conditioned on language goals. The commonly used 24-task protocol is static manipulation with language; navigation and composite tasks are left out."
    },
    "generalisation": {
     "value": [
      "object-instance",
      "visual",
      "object-pose"
     ],
     "display": "Evaluation uses only unseen object instances; two of five evaluation kitchens have styles never seen in training. Object placements are sampled at each reset.",
     "level": "inferred",
     "sources": [
      "s2",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "Object-instance and unseen styles are stated in the paper; mapping unseen styles to 'visual' and placement sampling to 'object-pose' is ours. Kitchen layouts are not held out: training demos come from random layouts.",
     "short": "New objects and new kitchen styles"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Real-robot experiments in the paper only test co-training with simulation data."
    },
    "simulator": {
     "value": "robosuite on MuJoCo",
     "display": "robosuite on MuJoCo (v0.1: robosuite robocasa_v0.1 branch; v0.2: robosuite v1.5). Optional NVIDIA Omniverse rendering.",
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Paper: about 25.2 steps per second with rendering, roughly real time.",
     "short": "robosuite, built on MuJoCo"
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda on an Omron mobile base (PandaOmron)",
     "display": "All experiments use a Franka Panda on an Omron mobile base. The framework also supports other mobile manipulators, humanoids and quadrupeds with arms.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "kitchen"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 100,
     "display": "100 tasks: 25 atomic and 75 composite (composite tasks proposed with LLM help). Common protocol: 24 atomic tasks (navigation excluded).",
     "level": "verified",
     "sources": [
      "s2",
      "s23",
      "s27"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "30 tasks with released datasets",
       "display": "25 atomic and 5 composite tasks have human datasets; 24 atomic tasks have MimicGen datasets (counted by us from the v0.2 registry)",
       "level": "inferred",
       "sources": [
        "s14",
        "s11"
       ]
      },
      {
       "value": "Composite results",
       "display": "The paper reports 5 composite tasks, single-task policies on 50 human demos: 0 to 2.0% from scratch, 0 to 12.0% with fine-tuning",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "100 tasks, of which 24 are in common use"
    },
    "scenes": {
     "value": 120,
     "display": "120 kitchen scenes: 10 floor plans x 12 styles. Evaluation uses 5 fixed scenes.",
     "level": "verified",
     "sources": [
      "s2",
      "s15",
      "s16",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The v0.2 code has 10 layout and 12 style files besides a 'playground' test file (our count). CONFLICT: the RoboCasa365 paper describes the original as having 100 scenes. The audit lists the five evaluation layout/style pairs as (1,1), (2,2), (4,4), (6,9), (7,10).",
     "short": "120 kitchens, of which 5 are used for testing"
    },
    "objects": {
     "value": 2509,
     "display": "2,509 objects in 153 categories, from Objaverse and Luma.ai text-to-3D",
     "level": "verified",
     "sources": [
      "s2",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Small conflict: the paper says 1,592 objects come from Luma.ai; the v0.2 docs table sums to 914 Objaverse and 1,595 AI-generated (our sum). Figure 1 says '2,500+'.",
     "short": "2,509 objects"
    },
    "demonstrations": {
     "value": 1500,
     "display": "1,500 human demonstrations by our arithmetic (50 per task for 25 atomic and 5 composite tasks) plus 100K+ MimicGen trajectories.",
     "level": "inferred",
     "sources": [
      "s2",
      "s11",
      "s14"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Human",
       "display": "1,250 for the 25 atomic tasks, collected by 4 operators with a SpaceMouse in random kitchens",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "MimicGen",
       "display": "72,000 trajectories (3,000 per task, 24 atomic tasks) with Objaverse objects, used in the paper, plus 28K with AI-generated objects",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Quality caveat",
       "display": "The paper says many generated trajectories show jerky motions and collisions",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "50 human demonstrations per task, and more than 100,000 generated ones"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "metric_detail": {
     "value": "24-task average success",
     "display": "Average success over the 24 atomic tasks, each tested on unseen objects in five fixed kitchens. Training data, trial count and checkpoint choice are not fixed.",
     "level": "verified",
     "sources": [
      "s2",
      "s23",
      "s27",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "The official policy-learning code is a robomimic branch with BC-Transformer. The paper's main result trains one multi-task BC-Transformer per dataset size and reports per-task and overall success.",
     "short": "Success averaged over 24 tasks"
    },
    "trials": {
     "value": "50 per task (paper)",
     "level": "verified",
     "sources": [
      "s2",
      "s23",
      "s24",
      "s27",
      "s28",
      "s29",
      "s31",
      "s19"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper",
       "display": "50 trials per task across 5 fixed scenes",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "GR00T N1",
       "display": "100 trials; maximum of the last 5 checkpoints",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "FLARE",
       "display": "50 episodes per task; maximum over the final 5 checkpoints",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "Cosmos Policy",
       "display": "50 trials per task x 3 seeds (3,600 trials)",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "World2Act",
       "display": "50 trials per task x 5 seeds",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "X-WAM",
       "display": "100 episodes per task",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "Z-1",
       "display": "48 rollouts per task (SFT), 64 (RL)",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "2026 audit probe",
       "display": "50 rollouts per task, 1,200 in total",
       "level": "verified",
       "sources": [
        "s19"
       ]
      }
     ],
     "short": "50 per task in the paper. Other papers use 48 to 150."
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s23",
      "s27",
      "s29",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "None of the papers we read gives a confidence interval or standard deviation for the 24-task average. The original paper gives mean and standard deviation only for its real-robot co-training results."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s5",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs the 2024 protocol. The official leaderboard covers RoboCasa365 only."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s5",
      "s17",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "No official leaderboard for the 2024 protocol. A community page (GINIGEN-AI, created 2026-06-29, not updated since) lists six matched-protocol entries copied from the RLDX-1 report."
    },
    "top_score": {
     "value": 79.2,
     "display": "79.2% average over 24 tasks (X-WAM, April 2026). Highest without RL fine-tuning that we found. Z-1 reaches 80.6% after RL in the simulator. Rows use different training data and trial counts and are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s29",
      "s31",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Each number is verified at its source. 'Highest' is our judgement over the audit tracker (snapshot 2026-05-21), the papers listed and a web search on 2026-10-10. Tracker rows above 79.2 (93.0, 89.3, 85.0, 81.33) use custom or partial task sets, per the tracker's own notes.",
     "items": [
      {
       "value": 28.8,
       "display": "BC-Transformer, Human-50, 2024-06: 28.8%",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "RoboCasa paper: 50 human demos per task, one multi-task policy, 50 trials per task",
       "data": {
        "model": "BC-Transformer, Human-50",
        "date": "2024-06",
        "avg": 28.8,
        "rl": false
       }
      },
      {
       "value": 47.6,
       "display": "BC-Transformer, Generated-3000, 2024-06: 47.6%",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "RoboCasa paper: 3,000 MimicGen demos per task",
       "data": {
        "model": "BC-Transformer, Generated-3000",
        "date": "2024-06",
        "avg": 47.6,
        "rl": false
       }
      },
      {
       "value": 57.3,
       "display": "DP-VLA, 2024-10: 57.3%",
       "level": "verified",
       "sources": [
        "s26"
       ],
       "note": "Generated-3000 data; OpenVLA-ft scored 9.8% in the same table",
       "data": {
        "model": "DP-VLA",
        "date": "2024-10",
        "avg": 57.3,
        "rl": false
       }
      },
      {
       "value": 49.6,
       "display": "GR00T-N1-2B, 2025-03: 49.6%",
       "level": "verified",
       "sources": [
        "s23"
       ],
       "note": "300 MimicGen demos per task (17.4% at 30, 32.1% at 100); 100 trials; best of last 5 checkpoints",
       "data": {
        "model": "GR00T-N1-2B",
        "date": "2025-03",
        "avg": 49.6,
        "rl": false
       }
      },
      {
       "value": 70.1,
       "display": "FLARE, 2025-05: 70.1%",
       "level": "verified",
       "sources": [
        "s24",
        "s27"
       ],
       "note": "Best of final 5 checkpoints, 50 episodes per task; 300 trajectories per task per its ablation figures (inferred). Cosmos Policy lists FLARE as 66.4",
       "data": {
        "model": "FLARE",
        "date": "2025-05",
        "avg": 70.1,
        "rl": false
       }
      },
      {
       "value": 66,
       "display": "Video Policy, 2025-08: 66.0%",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "note": "300 MimicGen demos per task; 63% with 50 human demos",
       "data": {
        "model": "Video Policy",
        "date": "2025-08",
        "avg": 66,
        "rl": false
       }
      },
      {
       "value": 67.1,
       "display": "Cosmos Policy, 2026-01: 67.1%",
       "level": "verified",
       "sources": [
        "s27"
       ],
       "note": "50 human demos per task; 3 seeds x 50 trials",
       "data": {
        "model": "Cosmos Policy",
        "date": "2026-01",
        "avg": 67.1,
        "rl": false
       }
      },
      {
       "value": 72.6,
       "display": "GR00T-N1.6-ft + World2Act, 2026-03: 72.6%",
       "level": "verified",
       "sources": [
        "s28"
       ],
       "note": "Base fine-tuned on 1,000 expert trajectories plus about 1,000 generated ones; 50 trials x 5 seeds",
       "data": {
        "model": "GR00T-N1.6-ft + World2Act",
        "date": "2026-03",
        "avg": 72.6,
        "rl": false
       }
      },
      {
       "value": 79.2,
       "display": "X-WAM, 2026-04: 79.2%",
       "level": "verified",
       "sources": [
        "s29"
       ],
       "note": "Pretraining includes 56,771 RoboCasa MimicGen episodes; 100 episodes per task",
       "data": {
        "model": "X-WAM",
        "date": "2026-04",
        "avg": 79.2,
        "rl": false
       }
      },
      {
       "value": 70.6,
       "display": "RLDX-1, 2026-05: 70.6%",
       "level": "verified",
       "sources": [
        "s30"
       ],
       "note": "300 MimicGen demos per task; 50 episodes per task",
       "data": {
        "model": "RLDX-1",
        "date": "2026-05",
        "avg": 70.6,
        "rl": false
       }
      },
      {
       "value": 70.8,
       "display": "GR00T N1.7, 2026-05: 70.8%",
       "level": "verified",
       "sources": [
        "s33"
       ],
       "note": "NVIDIA repository README (added 2026-05-26); training data from NVIDIA's simulation dataset; episode count not stated; N1.6 at 66.22",
       "data": {
        "model": "GR00T N1.7",
        "date": "2026-05",
        "avg": 70.8,
        "rl": false
       }
      },
      {
       "value": 80.6,
       "display": "Z-1 RL, 2026-06: 80.6%",
       "level": "verified",
       "sources": [
        "s31"
       ],
       "note": "SFT on 1,199 public human demos (67.4%), then RL fine-tuning in the simulator; 64 rollouts per task",
       "data": {
        "model": "Z-1 RL",
        "date": "2026-06",
        "avg": 80.6,
        "rl": true
       }
      }
     ],
     "short": "79.2% (April 2026), or 80.6% with reinforcement learning (RL) fine-tuning"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s7",
      "s5",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, copyright 2024 the RoboCasa Team, with a notice that partial MuJoCo code is under Apache-2.0. The GitHub API reports NOASSERTION because of the extra notice."
    },
    "license_data": {
     "value": "CC-BY-4.0",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "README licence section: 'Assets and Datasets: CC BY 4.0'."
    },
    "license_assets": {
     "value": "CC-BY-4.0",
     "display": "README: CC BY 4.0. Upstream terms of individual objects were not checked (items).",
     "level": "verified",
     "sources": [
      "s5",
      "s2",
      "s37",
      "s43"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Objaverse objects",
       "display": "Objaverse objects are individually licensed; its card lists 25K CC-BY-NC, 52K CC-BY-NC-SA and 16K CC-BY-SA objects among 800K+. RoboCasa does not say which licences its 914 to 917 Objaverse objects carry.",
       "level": "verified",
       "sources": [
        "s37",
        "s13"
       ]
      },
      {
       "value": "Fixtures and AI-generated objects",
       "display": "Fixtures come from 'online 3D model repositories' and most objects from Luma.ai; no per-source terms given",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "ManiSkill copy",
       "display": "ManiSkill's copy of RoboCasa scenes on Hugging Face is labelled MIT",
       "level": "verified",
       "sources": [
        "s43"
       ]
      }
     ],
     "short": "CC BY 4.0. The terms of the original object sources were not checked."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s14",
      "s44",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "v0.2 assets (about 5 GB) and datasets download from UT Box links in the code without registration. The links answered HTTP 200 to GET requests on 2026-10-10; HEAD requests return 404, so HEAD-only link checks wrongly report them dead.",
     "short": "Open, with downloads from UT Box"
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s7",
      "s5",
      "s37"
     ],
     "checked": "2026-10-10",
     "note": "MIT code and CC BY 4.0 assets and data allow commercial use with attribution. Some Objaverse objects carry non-commercial or share-alike terms, and RoboCasa does not say whether its objects include them. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Real-robot tests only show that RoboCasa data helps real training. No study compares RoboCasa scores with real-robot scores for the same policies.",
     "level": "inferred",
     "sources": [
      "s2",
      "s16",
      "s19",
      "s42"
     ],
     "checked": "2026-10-10",
     "note": "Paper: a Franka on DROID hardware, 3 pick-and-place tasks, 50 real demos each; co-training with all MimicGen data raised average success from 13.6% to 24.4% on seen objects and 2.6% to 9.3% on unseen objects (3 seeds). Simulation and real controllers differ (operational space control at 20 Hz vs 15 Hz). RoboCasa365 reports a similar co-training result. The 2026 audit calls a sim-vs-real ranking test impractical and does not run one. SureSim (2025-10) uses RoboCasa objects in its own ManiSkill3 twin, not RoboCasa tasks.",
     "short": "RoboCasa data helps train real robots. Its scores have not been compared with real-robot scores."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation."
    },
    "audits": {
     "value": "Audited in 2026",
     "display": "The 2026 audit 'What Are We Actually Benchmarking in Robot Manipulation?' covers the 24-task protocol, which it calls 'RSS24'. RoboCasa fails fewer diagnostics than LIBERO, CALVIN and SimplerEnv.",
     "level": "verified",
     "sources": [
      "s19",
      "s20",
      "s22"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Shortcut probe: none found",
       "display": "A 0.09B DINOv2+MLP probe, one model per task on 50 human demos, scored 18.8% (226/1,200) vs 49.6% (GR00T-N1) and 79.2% (X-WAM, best reported)",
       "level": "verified",
       "sources": [
        "s19"
       ],
       "note": "The audit's comparison rows use different training data: the probe had 50 human demos, GR00T-N1 300 generated demos."
      },
      {
       "value": "Significance: 16 of 30",
       "display": "Of 30 previous-best-to-new claims, 16 (53.3%) are provably significant, 10 inconclusive, 4 show no improvement, 0 provably not significant",
       "level": "verified",
       "sources": [
        "s19",
        "s22"
       ]
      },
      {
       "value": "Data-source dependence: not applicable",
       "display": "The audit says the training data covers the test, so this diagnostic does not apply",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "Bitwise reproducibility: fails",
       "display": "StarVLA rollouts diverged when only the CPU or only the GPU changed (issues.i4)",
       "level": "verified",
       "sources": [
        "s19"
       ]
      }
     ],
     "short": "A 2026 audit found no shortcut (a way to score well without the intended skill)."
    },
    "derived_benchmarks": {
     "value": [
      "RoboCasa365",
      "RoboCasa GR-1 Tabletop tasks"
     ],
     "display": "Benchmarks built on the RoboCasa framework",
     "level": "verified",
     "sources": [
      "s16",
      "s23"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RoboCasa365",
       "display": "2026: successor by the same lab with an official leaderboard",
       "level": "verified",
       "sources": [
        "s16",
        "s17"
       ]
      },
      {
       "value": "RoboCasa GR-1 Tabletop tasks",
       "display": "2025: 24 humanoid tabletop tasks that GR00T N1 says it built under the RoboCasa framework",
       "level": "verified",
       "sources": [
        "s23"
       ]
      },
      {
       "value": "Scenes in ManiSkill3",
       "display": "ManiSkill3 loads RoboCasa kitchens (RoboCasaKitchen-v1), without scored tasks",
       "level": "verified",
       "sources": [
        "s45"
       ]
      }
     ]
    },
    "citations": {
     "value": 575,
     "display": "575 (Semantic Scholar; 81 influential)",
     "level": "verified",
     "sources": [
      "s36"
     ],
     "checked": "2026-10-10",
     "short": "575"
    },
    "github_stars": {
     "value": 1796,
     "display": "1,796 stars, 265 forks (robocasa/robocasa)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The repository now hosts RoboCasa365, so the counts are shared.",
     "short": "1,796 (shared with RoboCasa365)"
    },
    "used_by": {
     "value": "69 papers reported results on the 24-task kitchen protocol by 2026-05-21; 97 on any RoboCasa protocol.",
     "level": "inferred",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Counted by us from the audit's released tracker CSV: 97 rows classified 'reports-results'; 69 include the rss24 protocol, 30 the GR-1 suite, 3 RoboCasa365. The audit marks 33 of the 69 as protocol-comparable. The audit calls its counts lower bounds.",
     "items": [
      {
       "value": "GR00T N1, N1.5, N1.6, N1.7",
       "display": "NVIDIA, 2025 to 2026 (paper and Isaac-GR00T README)",
       "level": "verified",
       "sources": [
        "s23",
        "s33"
       ]
      },
      {
       "value": "FLARE",
       "display": "NVIDIA, 2025-05",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "Video Policy",
       "display": "Columbia and Toyota Research Institute, 2025-08",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Cosmos Policy",
       "display": "NVIDIA and Stanford, 2026-01",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "World2Act",
       "display": "MBZUAI, 2026-03",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "X-WAM",
       "display": "Tsinghua, Xiaomi Robotics and others, 2026-04",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "Being-H0.7",
       "display": "BeingBeyond, 2026-04",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "RLDX-1",
       "display": "RLWRLD and KAIST, 2026-05",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "Z-1",
       "display": "Zioneer, 2026-06",
       "level": "verified",
       "sources": [
        "s31"
       ]
      }
     ],
     "short": "69 papers on the 24-task protocol (May 2026)"
    },
    "industry_use": {
     "value": [
      "NVIDIA",
      "Toyota Research Institute",
      "Xiaomi",
      "RLWRLD",
      "BeingBeyond",
      "Zioneer",
      "Hello Robot"
     ],
     "level": "verified",
     "sources": [
      "s23",
      "s33",
      "s25",
      "s29",
      "s30",
      "s32",
      "s31",
      "s41"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "NVIDIA",
       "display": "Co-developer; GR00T models, FLARE and Cosmos Policy report RoboCasa; Isaac-GR00T ships an evaluation example",
       "level": "verified",
       "sources": [
        "s2",
        "s23",
        "s33",
        "s27"
       ]
      },
      {
       "value": "Toyota Research Institute",
       "display": "Co-authors of Video Policy",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Xiaomi",
       "display": "Xiaomi Robotics co-authors X-WAM",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "RLWRLD",
       "display": "RLDX-1 technical report",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "BeingBeyond",
       "display": "Being-H0.7 report",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "Zioneer",
       "display": "Z-1 report",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "Hello Robot",
       "display": "stretch_mujoco loads RoboCasa kitchens for the Stretch robot (scenes only, not the benchmark)",
       "level": "verified",
       "sources": [
        "s41"
       ]
      }
     ]
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "Training data and trials differ between papers",
     "text": "Papers reporting the 24-task average train on 50 human demos per task (Cosmos Policy, Video Policy, the audit's probe), 300 MimicGen demos (GR00T N1, RLDX-1, Video Policy), 1,000 (World2Act's base) or 3,000 (the paper, DP-VLA). X-WAM pretrains on 56,771 RoboCasa MimicGen episodes before fine-tuning. Trials range from 48 to 150 per task. GR00T N1 and FLARE report the best of the last five checkpoints, each scored on the test scenes; there is no validation split. Z-1 adds RL fine-tuning inside the simulator. The same model can move by 20 points with data alone: GR00T-N1 scores 17.4%, 32.1% and 49.6% with 30, 100 and 300 demos.",
     "short": "Papers train on 50 to 3,000 demonstrations per task. Their averages cannot be compared directly.",
     "level": "verified",
     "sources": [
      "s2",
      "s23",
      "s24",
      "s25",
      "s26",
      "s27",
      "s28",
      "s29",
      "s30",
      "s31"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same model gets different numbers in different papers",
     "text": "GR00T N1.5 appears as 64.1 (Cosmos Policy, X-WAM), 65.7 (RLDX-1) and 59.7 (Z-1). Cosmos Policy reports 67.1 for itself; World2Act lists it at 65.7. FLARE reports 70.1; Cosmos Policy lists it at 66.4. pi0 appears as 62.5 with 300 demos (Cosmos Policy, X-WAM, RLDX-1) and 42.4 in Being-H0.7's 50-demo column; GR00T N1.6 as 66.22 (NVIDIA README) and 36.0 (Being-H0.7).",
     "short": "GR00T N1.5 is reported as 59.7, 64.1 and 65.7 in different papers.",
     "level": "verified",
     "sources": [
      "s27",
      "s29",
      "s30",
      "s31",
      "s28",
      "s24",
      "s32",
      "s33"
     ],
     "status": "open"
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "Results depend on the code version",
     "text": "The paper used v0.1, which has no tag. Datasets were recorded with robosuite 1.4.1 while v0.2 runs on robosuite 1.5, and users report version-mismatch warnings. A 2025-02-27 fix on the v0.2 branch changed how object rotations are sampled at reset. NVIDIA's GR00T evaluations run a fork 4 commits ahead of v0.2. Original task names also exist in RoboCasa365 code, where horizons were raised 1.5x in v1.0.1.",
     "short": "The same task names exist in v0.1, v0.2, later fixes, forks and the RoboCasa365 code.",
     "level": "verified",
     "sources": [
      "s10",
      "s39",
      "s9",
      "s34",
      "s35",
      "s6"
     ],
     "status": "open"
    },
    {
     "id": "i4",
     "type": "protocol-variance",
     "title": "Runs are not bit-for-bit repeatable across hardware",
     "text": "With the same StarVLA policy and settings, changing only the CPU made simulator and contact traces diverge at once, with images and actions diverging by step 4 and reward/success by step 19. Changing only the GPU led to reward/success divergence at step 18. Start states are drawn per reset, not per episode index; the maintainers say to pin them with set_ep_meta for paired comparisons.",
     "short": "Changing only the CPU or GPU changed episode outcomes.",
     "level": "verified",
     "sources": [
      "s19",
      "s38"
     ],
     "status": "open"
    },
    {
     "id": "i5",
     "type": "other",
     "title": "About half of claimed gains can be shown to be statistically significant",
     "text": "Of 30 previous-best-to-new comparisons on the 24-task protocol, 16 (53.3%) are provably significant at the 5% level from public scores, 10 are inconclusive and 4 show no improvement. The audit says the share is higher than LIBERO's 19.8% partly because RoboCasa has fewer reported results, so scores sit further apart.",
     "short": "16 of 30 claimed improvements can be shown to be statistically significant from the published scores.",
     "level": "verified",
     "sources": [
      "s19",
      "s22"
     ],
     "status": "open"
    },
    {
     "id": "i6",
     "type": "other",
     "title": "The asset licence may not match the terms of the original sources",
     "text": "RoboCasa licenses all assets under CC BY 4.0, but its Objaverse objects carry individual licences, and Objaverse includes non-commercial and share-alike objects. RoboCasa does not list per-object licences. ManiSkill's copy of the scenes is labelled MIT.",
     "short": "RoboCasa licenses its assets under CC BY 4.0. It does not publish the licence terms of the original objects.",
     "level": "inferred",
     "sources": [
      "s5",
      "s37",
      "s13",
      "s43"
     ],
     "status": "open",
     "note": "We did not check individual objects. Not legal advice."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A RoboCasa 24-task average can be compared only with another average from the same training data, the same number of trials, the same checkpoint rule and the same code. Published averages range from 28.8% to 80.6%, and the differences mostly reflect how much and what kind of training data was used. Read the training-data column before the score.",
     "short": "Check the training data before you compare RoboCasa averages.",
     "basis": [
      "facts.top_score",
      "issues.i1",
      "issues.i2",
      "issues.i3"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r2",
     "text": "The 2026 audit found no shortcut and a higher share of significant claims than on LIBERO. So a high RoboCasa score is better evidence of skill on these tasks than a high LIBERO score. It is still not evidence of real-world performance. The real-robot work tests whether simulation data helps training. It does not test whether scores predict real results.",
     "short": "RoboCasa scores hold up better than LIBERO scores in the 2026 audit. They have not been tested against real robots.",
     "basis": [
      "facts.audits",
      "facts.sim_to_real",
      "issues.i5"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r3",
     "text": "The common protocol uses only the 24 short atomic tasks. The 75 composite tasks, which contain the long, multi-step household chores, are rarely reported, and the original paper's composite results were at most 12%. A RoboCasa score therefore says little about multi-step chores.",
     "short": "Scores cover short atomic tasks only. They say little about multi-step chores.",
     "basis": [
      "facts.tasks",
      "facts.metric_detail"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r4",
     "text": "For new work, RoboCasa365 is the maintained successor, and it has an official leaderboard. The 2024 protocol is still useful for comparison with the large body of earlier results, if the code version is fixed.",
     "short": "New work should consider RoboCasa365. If you use the 2024 protocol, fix the code version.",
     "basis": [
      "facts.successor",
      "facts.status",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy (the robot's control model) will do on a real robot.",
     "sub": "No study has compared simulation and real-robot scores for the same policies.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy handles long, multi-step chores.",
     "sub": "The common protocol uses 24 short atomic tasks, each a single basic skill.",
     "basis": [
      "facts.tasks",
      "facts.metric_detail"
     ]
    },
    {
     "id": "l3",
     "text": "Whether scores from different papers can be compared directly.",
     "sub": "Papers train on 50 to 3,000 demonstrations per task.",
     "basis": [
      "issues.i1",
      "issues.i2"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "RoboCasa paper full text (Section V-C); RoboCasa365 paper; 2026 audit (2606.04233; no sim-vs-real test); PolaRiS (2512.16881; related-work mention only); 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (2606.10366; related work only); SureSim (2510.04354; uses RoboCasa objects in its own twin); Betting for Sim-to-Real (2604.24018), Robot Policy Evaluation for Sim-to-Real Transfer (2508.11117), Active Real-World Factor-Based Evaluation (2607.14439), Beyond Binary Success (2603.13616): no RoboCasa mention; audit tracker notes; web searches. No paired study found.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score (anything above 79.2% without RL)",
     "where": "Audit tracker (33 protocol-comparable rows), Cosmos Policy and X-WAM comparison tables, NVIDIA Isaac-GR00T README, community leaderboard, web search on 2026-10-10. Tracker rows above 79.2 use custom or partial task sets.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets (per-object terms)",
     "where": "README, paper, v0.2 docs objects page, Hugging Face asset card (no text). No per-object licence list found.",
     "date": "2026-10-10"
    },
    {
     "for": "uncertainty_reported",
     "where": "RoboCasa paper, GR00T N1, FLARE, Video Policy, Cosmos Policy, World2Act, X-WAM, RLDX-1, Z-1, DP-VLA result tables.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RoboCasa: Large-Scale Simulation of Everyday Tasks for Generalist Robots (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2406.02523",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RoboCasa paper, full text v1 (Sections III to V, Appendix VIII and IX, Figures 7, 10, 11, 13)",
     "url": "https://arxiv.org/pdf/2406.02523v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Robotics: Science and Systems XX, paper 50 (title 'RoboCasa: Large-Scale Simulation of Household Tasks for Generalist Robots')",
     "url": "https://roboticsproceedings.org/rss20/p050.html",
     "type": "paper",
     "publisher": "RSS 2024",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "GitHub API: robocasa/robocasa (stars, forks, created, licence field)",
     "url": "https://api.github.com/repos/robocasa/robocasa",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RoboCasa README at tag v0.2 (latest updates, licence section)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/README.md",
     "type": "repo",
     "publisher": "RoboCasa team (UT Austin)",
     "date": "2025-12-18",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RoboCasa README on main (RoboCasa365 updates, licence, citations)",
     "url": "https://github.com/robocasa/robocasa/blob/main/README.md",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2026-09-25",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "RoboCasa LICENSE at tag v0.2 (MIT, with Apache-2.0 notice for partial MuJoCo code)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/LICENSE",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "RoboCasa GitHub releases (v0.2 'Original RoboCasa Release', v1.0 'RoboCasa365 Release')",
     "url": "https://github.com/robocasa/robocasa/releases",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2026-02-18",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "RoboCasa v0.2 branch commit history (incl. placement-sampler fix f202a7eb, 2025-02-27)",
     "url": "https://github.com/robocasa/robocasa/commits/v0.2",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2025-12-18",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "RoboCasa issue #143 'Robocasa v0.1 where can I get it from?' (maintainer reply)",
     "url": "https://github.com/robocasa/robocasa/issues/143",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2025-05-12",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "RoboCasa v0.2 docs: Downloading Datasets",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/docs/use_cases/downloading_datasets.md",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "RoboCasa v0.2 docs: Policy Learning (robomimic 'robocasa' branch, BC-Transformer)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/docs/use_cases/policy_learning.md",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "RoboCasa v0.2 docs: Objects (per-category Objaverse and AI-generated counts)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/docs/tasks_scenes_assets/objects.md",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "RoboCasa v0.2 dataset registry (25 single-stage and 5 multi-stage task datasets; UT Box links)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/robocasa/utils/dataset_registry.py",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "RoboCasa v0.2 kitchen layout and style files",
     "url": "https://github.com/robocasa/robocasa/tree/v0.2/robocasa/models/assets/scenes",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "RoboCasa365: A Large-Scale Simulation Framework for Training and Benchmarking Generalist Robots",
     "url": "https://arxiv.org/abs/2603.04356",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026 per README citation)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "RoboCasa365 leaderboard page",
     "url": "https://robocasa.ai/leaderboard.html",
     "type": "leaderboard",
     "publisher": "RoboCasa team",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "RoboCasa project site (updates)",
     "url": "https://robocasa.ai/",
     "type": "site",
     "publisher": "RoboCasa team",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (Table 1, Figure 3, Appendix A.1, B, D)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Manipulation Benchmark Audit project page",
     "url": "https://ripl.github.io/manipulation_benchmark_audit/",
     "type": "site",
     "publisher": "TTIC RIPL (lists CoRL 2026)",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Audit artifacts: RoboCasa citation tracker CSV (snapshot 2026-05-21)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/HEAD/leaderboards/robocasa/robocasa_citation_tracker.csv",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Audit artifacts: statistical significance pie counts CSV",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/HEAD/analysis/release_current_values/statistical_significance_pie_counts.csv",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots (Table 4, evaluation protocol)",
     "url": "https://arxiv.org/abs/2503.14734",
     "type": "paper",
     "publisher": "arXiv (NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "FLARE: Robot Learning with Implicit World Modeling (Table 1)",
     "url": "https://arxiv.org/abs/2505.15659",
     "type": "paper",
     "publisher": "arXiv (NVIDIA and others)",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Video Generators are Robot Policies (Table 1)",
     "url": "https://arxiv.org/abs/2508.00795",
     "type": "paper",
     "publisher": "arXiv (Columbia University, Toyota Research Institute)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "A Dual Process VLA: Efficient Robotic Manipulation Leveraging VLM (DP-VLA, Table 1)",
     "url": "https://arxiv.org/abs/2410.15549",
     "type": "paper",
     "publisher": "arXiv (ETRI)",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning (Table 2)",
     "url": "https://arxiv.org/abs/2601.16163",
     "type": "paper",
     "publisher": "arXiv (NVIDIA, Stanford)",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "World2Act: Latent Action Post-Training from World Model Dynamics (Table 1)",
     "url": "https://arxiv.org/abs/2603.10422",
     "type": "paper",
     "publisher": "arXiv (MBZUAI)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Unified 4D World Action Modeling from Video Priors with Asynchronous Denoising (X-WAM, Table 1, Appendix B)",
     "url": "https://arxiv.org/abs/2604.26694",
     "type": "paper",
     "publisher": "arXiv (Tsinghua, Xiaomi Robotics, PKU, CASIA)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "RLDX-1 Technical Report (Table 1b, benchmark descriptions)",
     "url": "https://arxiv.org/abs/2605.03269",
     "type": "paper",
     "publisher": "arXiv (RLWRLD, KAIST)",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Z-1: Efficient Reinforcement Learning for Vision-Language-Action Models (Table 1, Appendix C.5)",
     "url": "https://arxiv.org/abs/2606.31846",
     "type": "paper",
     "publisher": "arXiv (Zioneer Robot Team)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Being-H0.7: A Latent World-Action Model from Egocentric Videos (Table 1)",
     "url": "https://arxiv.org/abs/2605.00078",
     "type": "paper",
     "publisher": "arXiv (BeingBeyond)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "NVIDIA Isaac-GR00T: RoboCasa evaluation example and checkpoint results (GR00T N1.6, N1.7)",
     "url": "https://github.com/NVIDIA/Isaac-GR00T/blob/main/examples/robocasa/README.md",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2026-05-26",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "NVIDIA Isaac-GR00T .gitmodules (robocasa submodule from squarefk/robocasa)",
     "url": "https://github.com/NVIDIA/Isaac-GR00T/blob/main/.gitmodules",
     "type": "repo",
     "publisher": "NVIDIA",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "GitHub compare robocasa v0.2 ... squarefk/robocasa@d89d481c (4 commits ahead, 1 behind)",
     "url": "https://github.com/robocasa/robocasa/compare/v0.2...squarefk:robocasa:d89d481ce9c76da7f179466981676e268aa842e5",
     "type": "repo",
     "publisher": "GitHub",
     "date": "2025-11-06",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "Semantic Scholar API record for arXiv:2406.02523",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2406.02523?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Objaverse dataset card (licence breakdown of individual objects)",
     "url": "https://huggingface.co/datasets/allenai/objaverse",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "RoboCasa issue #218 'Is there a supported way to hold the initial-state sequence fixed across two runs?'",
     "url": "https://github.com/robocasa/robocasa/issues/218",
     "type": "repo",
     "publisher": "RoboCasa team (maintainer reply)",
     "date": "2026-08-25",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "RoboCasa issue #144 'Potential Inconsistency in Dataset and Policy Learning Repo'",
     "url": "https://github.com/robocasa/robocasa/issues/144",
     "type": "repo",
     "publisher": "RoboCasa (user report)",
     "date": "2025-05-13",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "RoboCasa Kitchen Leaderboard (community Hugging Face Space by GINIGEN-AI)",
     "url": "https://ginigen-ai-robocasa-kitchen-leaderboard.hf.space/",
     "type": "secondary",
     "publisher": "GINIGEN-AI",
     "date": "2026-06-29",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "Hello Robot stretch_mujoco README (RoboCasa kitchen environments for Stretch)",
     "url": "https://github.com/hello-robot/stretch_mujoco",
     "type": "repo",
     "publisher": "Hello Robot",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "Reliable and Scalable Robot Policy Evaluation with Imperfect Simulators (SureSim; uses RoboCasa objects)",
     "url": "https://arxiv.org/abs/2510.04354",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "Hugging Face dataset card haosulab/RoboCasa (ManiSkill copy of RoboCasa scenes, labelled mit)",
     "url": "https://huggingface.co/datasets/haosulab/RoboCasa",
     "type": "repo",
     "publisher": "Hao Su lab",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "RoboCasa v0.2 asset download script (UT Box links)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/robocasa/scripts/download_kitchen_assets.py",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "ManiSkill documentation: Scene Datasets (RoboCasa scenes in ManiSkill3)",
     "url": "https://maniskill.readthedocs.io/en/latest/user_guide/datasets/scenes.html",
     "type": "site",
     "publisher": "mani-skill",
     "date": "2026",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created full entry from primary sources, starting from the checked basic entry and core-sim-a notes. Verified top scores at their sources (12 results), added the 2026 audit's tracker counts and significance breakdown, protocol and version issues, successor relation to RoboCasa365. Corrected: latest code change on v0.2 (2025-03, not only the 2024-10 README date); uncertainty reporting; access (Box links alive)."
    }
   ]
  },
  {
   "id": "robocasa365",
   "name": "RoboCasa365",
   "aliases": [
    "RoboCasa365: A Large-Scale Simulation Framework for Training and Benchmarking Generalist Robots",
    "RoboCasa v1.0",
    "RoboCasa 1.0.1",
    "RoboCasa365 Leaderboard"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed 50-task evaluation protocol with a public leaderboard that ranks robot policies by simulated success rate.",
   "summary": {
    "text": "Kitchen simulation benchmark with 365 tasks in 2,500 scenes; its leaderboard ranks policies on 50 target tasks.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "The University of Texas at Austin; NVIDIA Research",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2026-02",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub release v1.0 (RoboCasa365) 2026-02-18; arXiv v1 2026-03-04."
    },
    "latest_update": {
     "value": "v1.0.1 (2026-05-12): all task horizons increased 1.5x; 2026-07-07: per-frame subtask annotations for target composite datasets; leaderboard updated 2026-10-10",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "README 'Updates'; leaderboard page header."
    },
    "version": {
     "value": "1.0.1",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "setup.py and robocasa/__init__.py; no GitHub release object for 1.0.1."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "mobile-manipulation",
      "manipulation",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "220 of 365 tasks require mobile manipulation."
    },
    "embodiment": {
     "value": [
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Franka Panda with Omron mobile base for data collection; framework 'in principle' supports other mobile manipulators and humanoids."
    },
    "scene": {
     "value": [
      "kitchen"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "365 tasks (65 atomic, 300 composite); 2,500 pretraining kitchens + 10 target kitchens; 3,200+ objects; 30k human pretraining demos; 10k MimicGen demos per task for 60 atomic tasks; 25k target demos",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Weighting inferred: Axiom-0 (86.5*18 + 59.4*16 + 35.8*16)/50 = 61.6, matching the page."
    },
    "trials": {
     "value": "Conflict: 30 trials per task (paper App. G.2) vs 50 sampled scenarios per task (docs benchmarking overview)",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Paper: https://arxiv.org/pdf/2603.04356."
    },
    "evaluator": {
     "value": null,
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "display": "both: self-submitted JSON via pull request; RoboCasa team verifies (up to 10 days); closed models must give the team private access to checkpoint and eval code"
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Published 2026-04-06 (README). 16 models shown 2026-10-10; submissions folder holds 19 JSON files (our count)."
    },
    "top_score": {
     "value": "Axiom-0 (4Axiom Robotics, 2026-10-01): Overall 61.6 (Atomic-Seen 86.5%, Composite-Seen 59.4%, Composite-Unseen 35.8%); best open-source: Xiaomi-Robotics-1 57.4",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Copyright (c) 2026 the RoboCasa Team."
    },
    "license_data": {
     "value": "CC BY 4.0 (assets and datasets)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Hugging Face robocasa/robocasa-assets card: cc-by-4.0."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Upstream Objaverse object terms not checked. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Training-data evidence only: GR00T N1.5 mid-trained on 150 sim tasks then co-fine-tuned with 140 real demos reached 79.8% average real success vs 61.8% real-only (4 kitchen tasks, 20 trials each, DROID Panda; by the authors). No study pairing RoboCasa365 sim scores with real scores for several policies found."
    },
    "citations": {
     "value": 112,
     "display": "112 (Semantic Scholar, 2026-10-10)",
     "level": "verified",
     "sources": [
      "s12"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 1796,
     "display": "1796 (robocasa/robocasa, 2026-10-10)",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10"
    },
    "industry_use": {
     "value": "Leaderboard submitters include Xiaomi Robotics, Meta (ProWAM), TeleAI (PRTS), RLWRLD (RLDX-1), GigaAI, 4Axiom Robotics, Primotion, Phasor; NVIDIA GR00T N1.5/N1.6 and Physical Intelligence pi0/pi0.5 entered by the RoboCasa team",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboard updated 2026-10-10; commits 2026-09."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Top entry's submission says it is 'a full fine-tune of xiaomi-robotics-1' with paper_link ",
     "text": "Our reading of the submission JSON against the README rules; the organisers may have accepted it under the 'new recipe' clause.",
     "level": "inferred",
     "sources": [
      "s10"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "V1.0.1 lengthened horizons 1.5x; gr00t n1.5 re-evaluated (leaderboard 23.9 vs docs table 2",
     "text": "Docs table: https://raw.githubusercontent.com/robocasa/robocasa/main/docs/benchmarking/multitask_learning.md.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboCasa365: A Large-Scale Simulation Framework for Training and Benchmarking Generalist Robots",
     "url": "https://arxiv.org/pdf/2603.04356",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-03"
    },
    "s2": {
     "title": "robocasa/robocasa on GitHub (releases)",
     "url": "https://api.github.com/repos/robocasa/robocasa/releases",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "robocasa/robocasa on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/robocasa/robocasa/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "robocasa/robocasa on GitHub (file setup.py)",
     "url": "https://raw.githubusercontent.com/robocasa/robocasa/main/setup.py",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "RoboCasa365: A Large-Scale Simulation Framework for Training and Benchmarking Generalist Robots",
     "url": "https://arxiv.org/abs/2603.04356",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-03"
    },
    "s6": {
     "title": "robocasa/robocasa on GitHub (file datasets_overview.md)",
     "url": "https://raw.githubusercontent.com/robocasa/robocasa/main/docs/datasets/datasets_overview.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "RoboCasa Leaderboard",
     "url": "https://robocasa.ai/leaderboard.html",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "robocasa/robocasa on GitHub (file benchmarking_overview.md)",
     "url": "https://raw.githubusercontent.com/robocasa/robocasa/main/docs/benchmarking/benchmarking_overview.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "robocasa-benchmark/leaderboard on GitHub (repository)",
     "url": "https://github.com/robocasa-benchmark/leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "robocasa-benchmark/leaderboard on GitHub (file Axiom-0_2026_10_01.json)",
     "url": "https://github.com/robocasa-benchmark/leaderboard/blob/main/submissions/Axiom-0_2026_10_01.json",
     "type": "leaderboard",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s11": {
     "title": "robocasa/robocasa on GitHub (main)",
     "url": "https://raw.githubusercontent.com/robocasa/robocasa/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s12": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s13": {
     "title": "robocasa/robocasa on GitHub (repository)",
     "url": "https://api.github.com/repos/robocasa/robocasa",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "robochallenge",
   "name": "RoboChallenge",
   "full_name": "RoboChallenge: Large-scale Real-robot Evaluation of Embodied Policies",
   "aliases": [
    "Table30",
    "Table 30",
    "Table30 v2",
    "Table 30 V2",
    "RoboChallenge Table30"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "summary": {
    "text": "RoboChallenge is an online real-robot test service run by Dexmal with Hugging Face. Users run their policy on their own computers and control robots in Dexmal's lab over the internet, while staff set up and score each test, on two 30-task benchmarks: Table30 (2025-10 to 2026-05) and Table30 v2 (2026).",
    "short": "RoboChallenge tests robot policies (the models that control a robot) on real robots in Dexmal's lab. Users control the robots over the internet, and lab staff set up and score each test on 30 tabletop tasks.",
    "sources": [
     "s2",
     "s3",
     "s6"
    ]
   },
   "facts": {
    "kind": {
     "value": "arena",
     "level": "inferred",
     "sources": [
      "s2",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas. RoboChallenge runs the robots and scores the runs; the submitter runs the model on its own computers over the internet."
    },
    "kind_secondary": {
     "value": [
      "dataset",
      "challenge"
     ],
     "display": "Publishes demonstration data for every task, and hosts time-limited competitions (CVPR 2026 RoboChallenge Track, ICRA 2026 Dexmal whole-body-control track).",
     "level": "verified",
     "sources": [
      "s17",
      "s18",
      "s35",
      "s36"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "Table30 v2 (v2.0)",
     "display": "Two benchmark versions with separate leaderboards. Table30 (v1.0) ran from October 2025 and was retired on 2026-05-27. Table30 v2 (v2.0) opened as a preview for the CVPR 2026 competition in April 2026; its task data is dated 2026-05-28.",
     "level": "verified",
     "sources": [
      "s7",
      "s8",
      "s6",
      "s3"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Table30 v1.0",
       "display": "30 tasks on ALOHA (11), ARX5 (11), UR5 (6) and Franka (2); 19 single-arm and 11 two-arm. Task specs dated 2025-09-24. Retired 2026-05-27 after 'over 80,000 real-world robot task executions' and '20+ models'.",
       "level": "verified",
       "sources": [
        "s7",
        "s6"
       ]
      },
      {
       "value": "Table30 v2.0",
       "display": "30 tasks on DOS-W1 (10), ALOHA (10), ARX5 (7) and UR5 (3); 20 dual-arm. 18 new tasks plus 12 carried from v1 in revised form; Franka removed; multi-task models only; set-up alignment relaxed.",
       "level": "verified",
       "sources": [
        "s8",
        "s3",
        "s33"
       ],
       "note": "The V2 paper says DOS-W1 11 tasks and ALOHA 9; the site's task list says 10 and 10."
      },
      {
       "value": "v1 and v2 scores are not comparable",
       "display": "No task is identical across versions: 3 task names recur with changed props, steps or robot (for example three flowers in v1, four in v2). Robots, ranking rules and set-up rules also changed. We found no conversion between the two scales.",
       "level": "inferred",
       "sources": [
        "s7",
        "s8",
        "s3",
        "s14"
       ],
       "note": "Our reading of the two task lists and the V2 paper, which says the learning-testing protocol was 'fundamentally refactored'. Top success rate: 64.33% on v1, 40.67% on v2."
      },
      {
       "value": "Table 30 V2 Simulation (announced)",
       "display": "2026-08-19: a simulation copy of Table30 v2 in NVIDIA Isaac Lab, with Lightwheel and NVIDIA, aligned in tasks and scoring with the real benchmark. Not released as of 2026-10-10.",
       "level": "verified",
       "sources": [
        "s6",
        "s34"
       ]
      }
     ],
     "short": "Table30 v2 (2026). Version 1 was retired in May 2026. Scores from the two versions are not comparable."
    },
    "publishers": {
     "value": [
      "Dexmal",
      "Hugging Face"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s24"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Dexmal (原力灵机)",
       "display": "35 of 37 report authors and all 7 V2-paper authors. Supplies the robot cluster and staff. Legal name 北京原力灵机智能科技有限公司 (Beijing).",
       "level": "verified",
       "sources": [
        "s2",
        "s3",
        "s24",
        "s4"
       ]
      },
      {
       "value": "Hugging Face",
       "display": "2 of 37 report authors; named co-initiator.",
       "level": "verified",
       "sources": [
        "s2",
        "s4"
       ]
      },
      {
       "value": "RoboChallenge Committee (from 2025-11-20)",
       "display": "Partners named in the annual report: BAAI, AgiBot, Qwen, Galaxea, X Square Robot, Tsinghua University, Xi'an Jiaotong University, GOSIM.",
       "level": "verified",
       "sources": [
        "s4",
        "s5",
        "s16",
        "s22"
       ]
      }
     ],
     "note": "Report authors are listed in alphabetical order."
    },
    "builder_type": {
     "value": "robot-company",
     "level": "inferred",
     "sources": [
      "s24",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Dexmal's site lists its own robots (general-purpose, pallet-handling and research robots) and models (DM0, DM0.5)."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s24",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Dexmal is registered in Beijing. The annual report says 58.3% of active users were in China, 22.0% in the US."
    },
    "first_release": {
     "value": "2025-10",
     "display": "Site online 2025-10-17 (news item); the annual report gives 2025-10-15 as launch; arXiv report 2025-10-20.",
     "level": "verified",
     "sources": [
      "s6",
      "s4",
      "s1"
     ],
     "checked": "2026-10-10",
     "short": "October 2025"
    },
    "latest_update": {
     "value": "2026-10",
     "display": "Newest Table30 v2 run ended 2026-10-10; leaderboard files regenerated 2026-10-10 07:00 GMT. Last news item 2026-08-19.",
     "level": "verified",
     "sources": [
      "s13",
      "s10",
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "October 2026"
    },
    "status": {
     "value": "active",
     "display": "Table30 v2 runs every month since April 2026 (141, 298, 41, 162, 71, 289 and 17 public runs from April to October).",
     "level": "inferred",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "Active. Table30 v2 has had public runs every month since April 2026."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Task tags include precise3d, bimanual, repeated, temporal, softbody and classification."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "Test set-ups copy held-out demonstration episodes. v1 used a pixel-level image overlay; v2 relaxes this to rough alignment and assigns a random robot unit of the right type. Zero-shot object and background tracks are described for v2, but no such scores appear on the public board.",
     "level": "inferred",
     "sources": [
      "s2",
      "s3",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "The V2 paper says its competition preview does not score zero-shot or out-of-domain settings and that the out-of-domain mode will not be ranked.",
     "short": "Only the start positions change. Each test set-up copies a held-out demonstration."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "bimanual-arm",
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s8",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "v2: 20 dual-arm and 10 single-arm tasks. v1: 11 two-arm and 19 single-arm."
    },
    "robots": {
     "value": "ARX5, UR5, ALOHA, DOS-W1 (v2)",
     "display": "v2: ARX5, UR5, Cobot Magic ALOHA and DOS-W1 (a mobile two-arm robot with Airbot Play arms). v1: UR5 with Robotiq gripper, Franka (Robotiq gripper), Cobot Magic ALOHA and ARX-5. Intel RealSense cameras.",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "10 machines at launch (report); 20 by January 2026 (annual report)."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "'All tasks are executed on the table, or around a table.' One testing site."
    },
    "tasks": {
     "value": 30,
     "display": "30 tasks in each version; 3 task names recur, with changes.",
     "level": "verified",
     "sources": [
      "s7",
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "30 tasks in each version"
    },
    "demonstrations": {
     "value": 32980,
     "display": "v2: 32,980 demonstration episodes (1,001 to 1,675 per task), about 1.55 TB. v1: up to 1,000 per task; the site lists 25,627 episodes, about 1.57 TB.",
     "level": "verified",
     "sources": [
      "s8",
      "s7",
      "s17",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "The v2 total equals the sum of per-task counts. Reading v1's episode_number field as demonstrations is ours.",
     "short": "32,980 demonstrations in v2"
    },
    "scale": {
     "value": "about 50,000 public rollouts across both versions",
     "level": "inferred",
     "sources": [
      "s12",
      "s13"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v1 platform totals",
       "display": "41,969 rollouts, 209 test tokens issued and 82 developers submitting by January 2026 (annual report); 'over 80,000' robot task executions by 2026-05-27 (news).",
       "level": "verified",
       "sources": [
        "s4",
        "s6"
       ]
      },
      {
       "value": "v1 public run list",
       "display": "3,985 runs, 39,852 rollouts, 125 user names, 2025-09-20 to 2026-05-25.",
       "level": "inferred",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "v2 public run list",
       "display": "1,019 ranked runs, 10,189 rollouts, 51 user names, 2026-04-13 to 2026-10-10.",
       "level": "inferred",
       "sources": [
        "s13"
       ]
      },
      {
       "value": "Leaderboards",
       "display": "v1 final: 22 entries, all with 30 tasks. v2: 53 entries, 2 with all 30 tasks.",
       "level": "verified",
       "sources": [
        "s9",
        "s10"
       ]
      }
     ],
     "note": "Our counts. The public run lists hold fewer rollouts than the platform totals, so they appear to be a subset.",
     "short": "About 50,000 public test attempts across both versions"
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Ranked by success rate, then progress score."
    },
    "metric_detail": {
     "value": "average success rate over 30 tasks",
     "display": "Per task: success rate over 10 rollouts and a progress score (10 stage points per rollout, minus 0.5 per retry, so 0 to 100). Overall: plain averages over the 30 tasks; ranking by success rate, then progress score.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "v1's overall board lists only entries that completed all 30 tasks. v2 averages over all 30 tasks with untested tasks counted as 0, ranks partial entries, and uses the latest ranked run per task (issues.i5). The V2 paper announces rollout timeouts and a 'time to complete' score; no time field appears in the v2 leaderboard data on 2026-10-10.",
     "short": "Average success rate over 30 tasks. A progress score breaks ties."
    },
    "trials": {
     "value": "10 rollouts per task",
     "display": "10 rollouts per task per run; a full entry is 300 rollouts.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "10 attempts per task"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s10",
      "s9",
      "s32"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboard data give success rate and score only. PhAIL (2026-05) notes RoboChallenge reports no confidence intervals or paired tests."
    },
    "evaluator": {
     "value": "organiser-run",
     "display": "RoboChallenge staff set up and score every run on Dexmal's robots; the submitter runs the model on its own computers.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s3",
      "s15",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "Step by step: (1) Apply for a test token; download the task data from Hugging Face and fine-tune. (2) Submit a request naming the model and tasks; in v2 a ranked run must cover a whole robot group, and single-task 'test' runs are not ranked. (3) Staff queue the job (hours to days) and tell the user when to have the model running; in v2 the model must declare it is ready before it learns which task comes next. (4) For each rollout a tester places the props to match a reference image from a held-out demonstration and watches the run. (5) The user's program pulls camera images and robot state through RoboChallenge's API and pushes actions into a queue on the robot. (6) Testers score each stage by hand; scores get a second review and can be appealed. (7) Videos and logs are published."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s10",
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": 40.67,
     "display": "Table30 v2: VLA-DM0.5 (user 'KDDI Research', fine-tuned from Dexmal's DM0.5) 40.67% success, progress 54.42, all 30 tasks (read 2026-10-10). Table30 v1 final: Era0 (user 'Robotera') 64.33%, 76.34.",
     "level": "verified",
     "sources": [
      "s10",
      "s9",
      "s26"
     ],
     "checked": "2026-10-10",
     "note": "Read from leaderboard data last modified 2026-10-10 07:00 GMT. v1 and v2 numbers are not comparable. 'First place' claims refer to different dates and tracks (items). No chart data: two incomparable scales.",
     "items": [
      {
       "value": "v2 #1 VLA-DM0.5: 40.67% / 54.42",
       "display": "User 'KDDI Research'; 30 tasks; runs 2026-09-03 to 2026-09-16. Dexmal's OpenDM README says it was fine-tuned from DM0.5 and lists 43.0% success with the same 54.42 (issues.i3).",
       "level": "verified",
       "sources": [
        "s10",
        "s26"
       ]
      },
      {
       "value": "v2 #2 my16: 30.67% / 41.27",
       "display": "User 'Tymtbo'; 20 of 30 tasks tested, missing tasks count as 0.",
       "level": "verified",
       "sources": [
        "s10"
       ]
      },
      {
       "value": "v2 baseline: 14.33% / 31.48",
       "display": "RoboChallenge's multi-task baseline, 30 tasks, April 2026; sixth.",
       "level": "verified",
       "sources": [
        "s10"
       ]
      },
      {
       "value": "v1 final #1 Era0: 64.33% / 76.34",
       "display": "User 'Robotera'; task-specific models.",
       "level": "verified",
       "sources": [
        "s9"
       ]
      },
      {
       "value": "v1 #2 DM0 (Dexmal): 62.00% / 72.25",
       "display": "The operator's own model; its paper reports 62.0% (data before 2026-02-10).",
       "level": "verified",
       "sources": [
        "s9",
        "s25"
       ]
      },
      {
       "value": "v1 top multi-task: Lira_generalist 45.00% / 59.83",
       "display": "Qwen-RobotManip, submitted under an anonymous name; fifth overall.",
       "level": "verified",
       "sources": [
        "s9",
        "s28"
       ]
      },
      {
       "value": "v1 at launch: pi0.5 baseline 43.7% / 62.2",
       "display": "Technical report, October 2025; the leaderboard later shows 42.67% / 61.84.",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      },
      {
       "value": "Earlier first-place claims",
       "display": "Spirit-v1.5 '#1' as of 2026-01-11 (51.00% on the final board); GigaBrain-0.1 'first place' on 2026-02-09 (51.67%); the annual report gives 51% as the top on 2026-01-23.",
       "level": "verified",
       "sources": [
        "s29",
        "s30",
        "s4",
        "s9"
       ]
      }
     ],
     "short": "40.67% success on v2 (September 2026)"
    },
    "license_code": {
     "value": "CC-BY-NC-SA-4.0",
     "level": "verified",
     "sources": [
      "s20",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "RoboChallengeInference (client and mock server) LICENSE text is CC BY-NC-SA 4.0; GitHub's API shows NOASSERTION. The organisation's openpi fork is Apache-2.0; the Community repo has no licence. Dexmal's separate OpenDM client is Apache-2.0."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [
      "s17",
      "s18",
      "s19",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "No licence field or text in the Table30 and Table30v2 dataset cards; none of the 63 RoboChallenge datasets on Hugging Face has a licence tag (checked 2026-10-10). The governance document says datasets are to be published under open licences."
    },
    "license_assets": {
     "value": "not applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Physical robots and props; no 3D assets distributed."
    },
    "access": {
     "value": "application",
     "display": "Evaluation needs an approved account and test token (209 issued by January 2026). Data downloads are open, without registration.",
     "level": "verified",
     "sources": [
      "s4",
      "s15",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "Testing needs an approved application. The data is open to download."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s20",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "The official client is CC BY-NC-SA 4.0 and the data has no stated licence. Users can write their own client from the documented API. Not legal advice."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s2",
      "s6",
      "s34",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from real robots. A simulation copy of Table30 v2 (Isaac Lab, with Lightwheel and NVIDIA) was announced 2026-08-19; no comparison data yet. The annual report lists sim-versus-real comparison as a community request."
    },
    "real_reproducibility": {
     "value": "protocol-only",
     "display": "One testing site. Common robot models, published stage scoring and a reference-image set-up protocol, but no cross-site measurement.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s3",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "The tester study shows the person running the test changes results (validity v1). The annual report says lighting and even the tester's clothing changed a model's score between submissions. The V2 paper says camera and arm parameters cannot be reproduced exactly across robot units. In the public v2 run list, 146 consecutive pairs of ranked runs with the same user, model name and task differ by a median of 10 points of success rate (mean 16.8; 22.6% differ by 30 points or more); some pairs may involve changed models (our count)."
    },
    "published_at": {
     "value": "arXiv report; Table30 V2 at CVPR 2026 Workshops",
     "display": "Technical report arXiv 2510.17950 (single version). Table30 V2 paper: CVPR 2026 Workshops (GigaBrain Challenge), pp. 4461-4467.",
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "citations": {
     "value": 34,
     "display": "34 (Semantic Scholar; 2 influential)",
     "level": "verified",
     "sources": [
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "Technical report only; read 2026-10-11 03:48 UTC.",
     "short": "34"
    },
    "github_stars": {
     "value": 157,
     "display": "157 stars, 9 forks (RoboChallenge/RoboChallengeInference)",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "short": "157"
    },
    "dataset_downloads": {
     "value": 52088,
     "display": "Table30: 52,088 all time (3,752 'downloads'); Table30v2: 47,856 all time (2,822)",
     "level": "verified",
     "sources": [
      "s17",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Hub API fields; we did not check the window behind 'downloads'. The annual report gave 17K for Table30 in January 2026.",
     "short": "52,088 for Table30"
    },
    "used_by": {
     "value": "At least 6 model reports publish RoboChallenge results (2026).",
     "level": "verified",
     "sources": [
      "s25",
      "s28",
      "s31",
      "s29",
      "s30",
      "s26"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "DM0 (Dexmal and StepFun, 2026-02)",
       "display": "Table30: 62.0% specialist, 37.3% generalist.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "Spirit-v1.5 (Spirit AI, 2026-01)",
       "display": "Claims '#1' on Table30 as of 2026-01-11; repository includes the RoboChallenge runtime.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "GigaBrain-0.1 (GigaAI, 2026-02)",
       "display": "News item: first place on 2026-02-09.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "StarVLA-alpha (2026-04)",
       "display": "Generalist results on Table30; says it beats pi0.5 by 20%.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "Qwen-RobotManip (Qwen, 2026-06)",
       "display": "First in the v1 generalist track as 'Lira_generalist' (45% / 59.83).",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "DM0.5 (Dexmal) / VLA-DM0.5 (KDDI Research, 2026-10)",
       "display": "Dexmal released DM0.5 Table30 v2 checkpoints in August 2026; a KDDI fine-tune leads v2.",
       "level": "verified",
       "sources": [
        "s26",
        "s27"
       ]
      },
      {
       "value": "Training data reused",
       "display": "Wall-OSS-0.5 (X Square Robot) and WSA1 list RoboChallenge data in their pretraining mix.",
       "level": "verified",
       "sources": [
        "s37",
        "s38"
       ]
      }
     ],
     "short": "At least 6 model reports (2026)"
    },
    "industry_use": {
     "value": [
      "Dexmal",
      "Hugging Face",
      "Qwen",
      "Spirit AI",
      "GigaAI",
      "X Square Robot",
      "Robotera",
      "KDDI Research",
      "NVIDIA",
      "Lightwheel"
     ],
     "level": "verified",
     "sources": [
      "s4",
      "s9",
      "s10",
      "s25",
      "s28",
      "s29",
      "s30",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "Organisation names behind leaderboard user names are as self-reported or as stated in the cited reports. NVIDIA and Lightwheel join through the simulation counterpart; AgiBot co-sponsors the ICRA 2026 competition."
    },
    "derived_benchmarks": {
     "value": [
      "CVPR 2026 RoboChallenge Track",
      "Table 30 V2 Simulation (announced)"
     ],
     "level": "verified",
     "sources": [
      "s35",
      "s34",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "CVPR 2026 RoboChallenge Track",
       "display": "Part of GigaBrain Challenge 2026; the 'Table30 CVPR version' uses the 30 v2 task IDs; single model, multiple tasks.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "Table 30 V2 Simulation",
       "display": "Announced 2026-08-19 in Isaac Lab with Lightwheel and NVIDIA.",
       "level": "verified",
       "sources": [
        "s34",
        "s6"
       ]
      }
     ]
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "The operator cannot check which model ran",
     "text": "Inference runs on the submitter's computers. The report says the platform has no means to check that the model actually run matches the claimed one; a user could use per-task models when a generalist is expected, or even intervene by hand. The annual report repeats this. For v2 the operator added a timing check (the model must declare it is ready before it learns the task) and calls it necessary but not sufficient; enforcement rests on users accepting the rules. Entries can use anonymous names: Qwen submitted as 'Lira_generalist'.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s3",
      "s28"
     ],
     "status": "open",
     "short": "Models run on the submitters' own computers. The organisers cannot verify which model was actually used."
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "The person who sets up the test changes the score",
     "text": "In the report's tester study, the same model scored 10%, 0% and 30% on 'pour french fries' and 50%, 70% and 80% on 'stack bowls' with experienced, first-time and adaptive (model-author) testers. Adaptive testers placed objects in 'sweet spots'. The annual report says lighting and the tester's clothing changed a model's score between two submissions. The V2 paper says camera and arm parameters differ between robot units.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s3"
     ],
     "status": "open",
     "mitigation": {
      "text": "v1 used a reference-image overlay so testers copy held-out demonstration start states. v2 relaxed this to rough alignment for speed and assigns robot units at random.",
      "sources": [
       "s2",
       "s3"
      ]
     },
     "short": "In the RoboChallenge report's study, changing the person who set up the test changed one model's success rate by up to 30 points."
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "The same entries have different numbers in different sources",
     "text": "The report's pi0.5 baseline is 43.7% / 62.2; the leaderboard shows 42.67% / 61.84, with two tasks different (arrange fruits 80% to 40%, sort electronic products 40% to 50%). pi0's progress score is 47.6 in the report and 46.41 on the board. Dexmal's OpenDM README gives VLA-DM0.5 43.0% success; the board shows 40.67% with the same 54.42 score. Qwen's report cites DM0_generalist at 48.43; the board shows 49.08. The V2 paper and the site differ on tasks per robot. In December 2025 the site warned that some displayed results were temporary, partial or for debugging.",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s11",
      "s26",
      "s10",
      "s28",
      "s3",
      "s8",
      "s6"
     ],
     "status": "open",
     "short": "The RoboChallenge report, the leaderboard and the model makers' own reports give different numbers for the same entries."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The operator's own models compete on its leaderboard",
     "text": "Dexmal staff run the robots, set-ups and scoring, and Dexmal's DM0 ranks second on the final v1 board (62.00%); DM0_generalist is eighth. The current v2 leader, VLA-DM0.5, is a fine-tune of Dexmal's DM0.5, whose Table30 v2 checkpoints Dexmal released in August 2026. Several committee partners also have entries.",
     "level": "inferred",
     "sources": [
      "s9",
      "s10",
      "s25",
      "s26",
      "s27",
      "s4"
     ],
     "status": "open",
     "note": "A structural conflict of interest; we found no evidence of unequal treatment.",
     "short": "Dexmal, which runs and scores the tests, also has models near the top of the leaderboard."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "Version 2 ranking rules reward re-runs and rank incomplete entries",
     "text": "In v2, the latest ranked run of each task replaces earlier ones (all 46 tasks with several ranked runs match the latest run, only 21 match the best), so a team can re-run a task and stop when satisfied. Untested tasks count as 0 and partial entries are ranked (second place has 20 of 30 tasks). v1's overall board required all 30 tasks.",
     "level": "inferred",
     "sources": [
      "s10",
      "s13",
      "s14",
      "s15",
      "s4"
     ],
     "status": "open",
     "note": "Rules worked out by us from the public leaderboard and run data on 2026-10-10.",
     "short": "In version 2, only the newest run of each task counts, so a team can re-run a task until it is satisfied. Entries with untested tasks are still ranked."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Scores from 10 attempts per task are shown without error bars",
     "text": "Scores rest on 10 rollouts per task and are published without intervals. PhAIL notes that neither RoboArena nor RoboChallenge reports confidence intervals or paired tests. Repeated ranked v2 runs of the same entry and task differ by a median of 10 points (our count).",
     "level": "verified",
     "sources": [
      "s32",
      "s10",
      "s13"
     ],
     "status": "open",
     "short": "Each score is a single number from 10 attempts per task. No measure of uncertainty is shown."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Licences do not match the stated policy",
     "text": "The committee's governance model says results, datasets and code are to be published under open licences. The datasets carry no licence and the official client is non-commercial (CC BY-NC-SA 4.0).",
     "level": "inferred",
     "sources": [
      "s16",
      "s19",
      "s20"
     ],
     "status": "open",
     "short": "The committee's policy says data and code are published under open licences. The datasets have no licence, and the official client software is licensed for non-commercial use only."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A Table30 score shows how well a policy, fine-tuned on RoboChallenge's own demonstrations, repeats those tasks on the same robots in the same lab. It does not show performance in new places or with new objects. The v2 zero-shot tracks, which test new objects and backgrounds, are not on the public leaderboard yet.",
     "basis": [
      "facts.generalisation",
      "facts.demonstrations",
      "facts.real_reproducibility"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "A score shows how well a policy repeats known tasks in one lab."
    },
    {
     "id": "r2",
     "text": "Do not compare version 1 and version 2 numbers, and do not read the drop from 64% to 41% as a loss of skill. Tasks, robots and rules all changed between the versions.",
     "basis": [
      "facts.version",
      "facts.top_score"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Do not compare scores from version 1 and version 2."
    },
    {
     "id": "r3",
     "text": "Treat leaderboard places as claims by the submitters, checked by human scoring. The model runs on the submitter's side, entry names can be anonymous, and the newest run counts. The operator's own model family is near the top of both versions.",
     "basis": [
      "issues.i1",
      "issues.i4",
      "issues.i5"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Leaderboard places are claims by the submitters, checked only by human scoring of each attempt."
    },
    {
     "id": "r4",
     "text": "With 10 attempts per task and no error bars, gaps of a few points between entries are within the normal variation between runs. Many first-place claims refer to different dates or tracks.",
     "basis": [
      "facts.trials",
      "issues.i6",
      "facts.top_score"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Do not read much into gaps of a few points. Check the date and track behind any first-place claim."
    }
   ],
   "searched": [
    {
     "for": "license_data",
     "where": "Table30 and Table30v2 dataset cards, Hugging Face API for all 63 RoboChallenge datasets, governance PDF, annual reports, site code.",
     "date": "2026-10-10"
    },
    {
     "for": "cross-site or independent checks",
     "where": "Report, V2 paper, annual reports (Chinese and English), Semantic Scholar list of 34 citing papers (full texts of 10 scanned), web search for critiques in English and Chinese. Found only PhAIL's remark on uncertainty.",
     "date": "2026-10-10"
    },
    {
     "for": "zero-shot and time-to-complete scores",
     "where": "v2 leaderboard and run data, benchmark metadata, site code. Not present on 2026-10-10.",
     "date": "2026-10-10"
    },
    {
     "for": "official Chinese announcement of v2",
     "where": "Site news, Community README, web search for the WeChat original; only the QbitAI reprint (secondary) was found. The CVPR 2026 Workshops paper is used instead.",
     "date": "2026-10-10"
    },
    {
     "for": "statement on v1/v2 comparability",
     "where": "V2 paper, site news, benchmark metadata. No explicit statement; our conclusion is inferred.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RoboChallenge: Large-scale Real-robot Evaluation of Embodied Policies (arXiv abstract page, single version v1)",
     "url": "https://arxiv.org/abs/2510.17950",
     "type": "paper",
     "publisher": "arXiv (Dexmal, Hugging Face)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RoboChallenge technical report, full text v1 (Sections 2-4, Appendix A; Figure 3 tester study)",
     "url": "https://arxiv.org/html/2510.17950v1",
     "type": "paper",
     "publisher": "arXiv (Dexmal, Hugging Face)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Table30 V2: Evaluating Generalized Models by Real Robots at Scale (CVPR 2026 Workshops, pp. 4461-4467)",
     "url": "https://openaccess.thecvf.com/content/CVPR2026W/GigaBrainChallenge/papers/Ma_Table30_V2_Evaluating_Generalized_Models_by_Real_Robots_at_Scale_CVPRW_2026_paper.pdf",
     "type": "paper",
     "publisher": "CVF Open Access (Dexmal authors)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "2025 RoboChallenge 年度报告 (Annual Report 2025 Q4 - 2026 Q1, Chinese edition)",
     "url": "https://robochallenge.ai/2025%20RoboChallenge%20%E5%B9%B4%E5%BA%A6%E6%8A%A5%E5%91%8A.pdf",
     "type": "report",
     "publisher": "RoboChallenge Committee",
     "date": "2026-01-30",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "2025 RoboChallenge Annual Report (English edition)",
     "url": "https://robochallenge.ai/2025%20RoboChallenge%20Annual%20Report.pdf",
     "type": "report",
     "publisher": "RoboChallenge Committee",
     "date": "2026-01-30",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RoboChallenge news page (dated items 2025-10-17 to 2026-08-19; read from the site's page code)",
     "url": "https://robochallenge.ai/assets/news-083b8373.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026-08-19",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "Table 30 (v1.0) benchmark metadata: tasks, robots, scoring stages",
     "url": "https://robochallenge.ai/api/v1/benchmark/benchmark_list.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Table 30 V2 (v2.0) benchmark metadata: tasks, robots, prompts, scoring stages, episode counts",
     "url": "https://robochallenge.ai/api/v2/benchmark/benchmark_list.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Table 30 (v1) final leaderboard data (22 entries)",
     "url": "https://robochallenge.ai/api/v1/leaderboard/leaderboard_all.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "Table 30 V2 leaderboard data (53 entries, per-task results)",
     "url": "https://robochallenge.ai/api/v2/leaderboard/leaderboard_table30_v2.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Table 30 (v1) per-task leaderboard data",
     "url": "https://robochallenge.ai/api/v1/leaderboard/leaderboard_task_all.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Table 30 (v1) public run list (3,985 runs with rollout scores)",
     "url": "https://robochallenge.ai/api/v1/runs/runs_list.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Table 30 V2 public run list (1,019 ranked runs with rollout scores)",
     "url": "https://robochallenge.ai/api/v2/runs/runs_table30_v2.json",
     "type": "leaderboard",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "RoboChallenge leaderboard page code (ranking order; v2 entries treated as multi-task)",
     "url": "https://robochallenge.ai/assets/leaderboardStore-95e76d75.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "RoboChallenge v2 submission form code (ranked versus test runs; whole robot group required)",
     "url": "https://robochallenge.ai/assets/evaluateYourPolicyV2-4b57c9cb.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "RoboChallenge Committee Governance Model",
     "url": "https://robochallenge.ai/RoboChallenge_Committee_Governance_Model.pdf",
     "type": "site",
     "publisher": "RoboChallenge Committee",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "RoboChallenge/Table30 dataset card and Hub record",
     "url": "https://huggingface.co/datasets/RoboChallenge/Table30",
     "type": "repo",
     "publisher": "RoboChallenge",
     "date": "2026-01-13",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "RoboChallenge/Table30v2 dataset card and Hub record",
     "url": "https://huggingface.co/datasets/RoboChallenge/Table30v2",
     "type": "repo",
     "publisher": "RoboChallenge",
     "date": "2026-06-05",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Hugging Face API: all 63 RoboChallenge datasets (no licence tags)",
     "url": "https://huggingface.co/api/datasets?author=RoboChallenge&limit=300&full=true",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "RoboChallengeInference LICENSE (CC BY-NC-SA 4.0) and README",
     "url": "https://github.com/RoboChallenge/RoboChallengeInference/blob/main/LICENSE",
     "type": "repo",
     "publisher": "RoboChallenge",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "GitHub API: RoboChallenge organisation repositories (stars, licences)",
     "url": "https://api.github.com/orgs/RoboChallenge/repos",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "RoboChallenge Community README (Chinese edition)",
     "url": "https://github.com/RoboChallenge/Community/blob/main/README_zh-CN.md",
     "type": "repo",
     "publisher": "RoboChallenge",
     "date": "2026-01-30",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Semantic Scholar API record for arXiv:2510.17950",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2510.17950?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Dexmal official website (北京原力灵机智能科技有限公司; robots and models)",
     "url": "https://www.dexmal.com/",
     "type": "site",
     "publisher": "Dexmal",
     "date": "2026-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "DM0: An Embodied-Native Vision-Language-Action Model towards Physical AI (Section 4.2, RoboChallenge results)",
     "url": "https://arxiv.org/html/2602.14974v1",
     "type": "paper",
     "publisher": "arXiv (Dexmal, StepFun)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "Dexmal OpenDM README (news 2026-10-05: KDDI Research's VLA-DM0.5 tops Table30 V2; results table)",
     "url": "https://github.com/dexmal/opendm/blob/91ddc9a0ddb946170d6bac3382876043c24e931d/README.md",
     "type": "repo",
     "publisher": "Dexmal",
     "date": "2026-10-05",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "Dexmal/DM05-Table30v2-W1 model card (DM0.5 checkpoints for Table30 v2)",
     "url": "https://huggingface.co/Dexmal/DM05-Table30v2-W1",
     "type": "repo",
     "publisher": "Dexmal",
     "date": "2026-08-06",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "Qwen-RobotManip Technical Report (v2; Section 6.3.2, Table30-v1 generalist track)",
     "url": "https://arxiv.org/html/2606.17846v2",
     "type": "paper",
     "publisher": "arXiv (Qwen team)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "Spirit-v1.5 repository README ('ranks #1 on RoboChallenge Table30' as of 2026-01-11)",
     "url": "https://github.com/Spirit-AI-Team/spirit-v1.5",
     "type": "repo",
     "publisher": "Spirit AI",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "GigaBrain-0 repository README (news: GigaBrain-0.1 first place on RoboChallenge, 2026-02-09)",
     "url": "https://github.com/open-gigaai/giga-brain-0",
     "type": "repo",
     "publisher": "GigaAI",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "StarVLA-alpha: Reducing Complexity in Vision-Language-Action Systems (v2; Section 5, Appendix E)",
     "url": "https://arxiv.org/html/2604.11757v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "PhAIL: A Real-Robot VLA Benchmark and Distributional Methodology (Related Work)",
     "url": "https://arxiv.org/html/2605.29710",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "QbitAI: authorised reprint of RoboChallenge's Table30 V2 launch announcement",
     "url": "https://www.qbitai.com/2026/03/391744.html",
     "type": "secondary",
     "publisher": "QbitAI",
     "date": "2026-03-24",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "RoboChallenge Table 30 V2 simulation page (Lightwheel, NVIDIA Isaac Lab)",
     "url": "https://robochallenge.ai/assets/table30V2Simulation-c0b62a39.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026-08-19",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "RoboChallenge CVPR 2026 competition page and task list (same 30 task IDs as v2)",
     "url": "https://robochallenge.ai/assets/cvprTasks-68dbcace.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "RoboChallenge ICRA 2026 competition page (Dexmal WBC track, supermarket tasks)",
     "url": "https://robochallenge.ai/assets/icraCompetition-42b66875.js",
     "type": "site",
     "publisher": "RoboChallenge",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Wall-OSS-0.5 Technical Report (pretraining data includes RoboChallenge)",
     "url": "https://arxiv.org/html/2605.30877",
     "type": "paper",
     "publisher": "arXiv (X Square Robot)",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "WSA1: a 3D-Centric World-Spatial-Action Model (pretraining data includes RoboChallenge)",
     "url": "https://arxiv.org/html/2607.03941",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-07",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the checked basic entry. Added the Table30 V2 CVPRW paper, both annual reports, governance model, v1/v2 comparison, leaderboard rules from site code and data, run counts, model reports and tester study. Resolved the basic entry's issue on Qwen-RobotManip: its 'ranks 1st' claim refers to the v1 generalist track, where it is listed as Lira_generalist."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy does in new places or with new objects.",
     "sub": "The tests use the same tasks as the training demonstrations, and each test copies the set-up of a recorded demonstration.",
     "basis": [
      "facts.generalisation"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a policy would get the same score in another lab.",
     "sub": "All tests run at one site. No one has checked the results at a second site.",
     "basis": [
      "facts.real_reproducibility",
      "issues.i2"
     ]
    },
    {
     "id": "l3",
     "text": "Which model was actually tested.",
     "sub": "The model runs on the submitter's own computers, where the operator cannot check it.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l4",
     "text": "Whether policies got better between version 1 and version 2.",
     "sub": "Version 1 and version 2 use different tasks and rules.",
     "basis": [
      "facts.version"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "Tester-variation test (report Figure 3)",
     "date": "2025-10",
     "by": "authors",
     "method": "Two tasks were each run with one model by three kinds of tester: data collectors, first-time testers and the model's authors.",
     "result": "Success was 10%, 0% and 30% on pouring fries, and 50%, 70% and 80% on stacking bowls.",
     "authors_view": "varies considerably",
     "n_policies": 2,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Values read from the figure's bar labels; run counts not stated. This checks the test protocol, not agreement with another measure. No comparison of RoboChallenge scores with another real-world evaluation was found."
    }
   ]
  },
  {
   "id": "robodojo",
   "name": "RoboDojo",
   "aliases": [
    "RoboDojo-Sim",
    "RoboDojo-RealWorld",
    "RoboDojo-RealEval"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores generalist manipulation policies in closed-loop simulation and on real robots, with an organiser-run leaderboard.",
   "summary": {
    "text": "Bimanual manipulation benchmark with 42 simulated and 18 real-robot tasks and an organiser-run leaderboard.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "44 authors. Affiliations in paper: MMLab@HKU, UC Berkeley, THU, PKU, Stanford, MIT, UNC, Princeton, CMU, NUS, NTU, CUHK, IC, SJTU, NU, HKUST (GZ), ZJU, Yale. Leaderboard maintained by AI MMLab Club, a non-profit.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Lead group MMLab@HKU (Hong Kong); co-authors in North America, Europe and Asia. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2026-07 (arXiv v1 2026-07-05; README: paper and code released July 6, 2026)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "README news 2026-09-16/17: fixed a single-frame get_obs discrepancy, updated code and Hugging Face data, fixed RGB channel order with RoboTwin, updated swap_T assets. Leaderboard 'Updated 2026.10.9'. Latest commit 2026-10-08.",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "version": {
     "value": "No version tags (GitHub releases empty); paper arXiv v3 2026-07-08",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "arXiv preprint (no journal-ref)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim+real",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Sim dimensions: generalization, memory, precision, long-horizon, open-vocabulary instruction following. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "robots": {
     "value": "Sim: ARX X5 bimanual platform. Real: ARX X5, Piper, Piper X (bimanual).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Fixed bimanual workstations with controlled workspace; scene count not stated. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Reported as 'score / SR%'. Sim headline = mean across the five dimensions. Real: each trial scored by three double-blind evaluators, averaged; score includes sub-step completion. Checked 2026-10-10."
    },
    "trials": {
     "value": "Sim: 50 episodes per task in paper; leaderboard protocol 3 seeds (150 trials per task). Real: 10 trials per task (180 per policy).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "evaluator": {
     "value": null,
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "organiser-run: 'Reported scores are computed by the official evaluation system rather than self-reported by participants'; hidden verification layouts; checkpoint and code must be released for verified entries",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "display": "Official (. 2026-10-10: Sim board 57 models (updated 2026-10-09), top VPP2 (RobotEra) 39.26 / 32.26%; RealWorld board 11 models, top OpenWAM-alpha 37.60 / 24.40%, pi0.5 22.90 / 12.80%. Paper launch (2026-07-06): top sim Hy-Embodied-0.5-VLA 13.07 / 8.80%; human expert reference 80.42 / 76.03%.)",
     "note": "Rendered in a browser; page is JavaScript-only. Checked 2026-10-10."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. Checked 2026-10-10."
    },
    "license_data": {
     "value": "Apache-2.0 (Hugging Face dataset card YAML 'license: apache-2.0'; card has no other text). Sim assets hosted on ModelScope with no licence stated on the download page.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Download page: https://robodojo-benchmark.com/doc/usage/install-and-download/ Checked 2026-10-10."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "MIT code and Apache-2.0 data; asset licence unstated. Not legal advice. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Tried, not measured (: 10 policies evaluated in real world and sim on different task sets; authors describe 'partial but not complete alignment' (e.g. pi0.5 strongest real, among leaders in sim; InternVLA-A1 and GalaxeaVLA rank higher in real than in sim). No correlation statistic. Authors state the real benchmark does not measure direct sim-to-real transfer on matched tasks.)",
     "note": "10 policies run in both sim and real, but on different task sets (real tasks are not copies of sim tasks). Authors report only qualitative 'partial but not complete alignment' of rankings; no correlation statistic."
    },
    "real_reproducibility": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "protocol-only: standardized hardware, camera and lighting poses, reset procedure and remote cloud access; no cross-site measurement reported",
     "note": "Checked 2026-10-10."
    },
    "citations": {
     "value": 47,
     "display": "47",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 709,
     "display": "709",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "industry_use": {
     "value": "Leaderboard entries contributed by companies including RobotEra, Li Auto, Horizon Robotics, Xiaomi Robotics, Meituan Robotics, Galaxea AI, Dexmal, Tencent Robotics X (contributor column)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboDojo: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies",
     "url": "https://arxiv.org/abs/2607.04434",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-07"
    },
    "s2": {
     "title": "RoboDojo: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies (full text)",
     "url": "https://arxiv.org/html/2607.04434v3",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-07"
    },
    "s3": {
     "title": "RoboDojo-Benchmark/RoboDojo on GitHub (repository)",
     "url": "https://github.com/RoboDojo-Benchmark/RoboDojo",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "RoboDojo: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies",
     "url": "https://robodojo-benchmark.com/leaderboard/protocol",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RoboDojo: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies",
     "url": "https://robodojo-benchmark.com/leaderboard",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RoboDojo-Benchmark/RoboDojo on GitHub (blob)",
     "url": "https://github.com/RoboDojo-Benchmark/RoboDojo/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s7": {
     "title": "RoboDojo-Benchmark/RoboDojo on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/RoboDojo-Benchmark/RoboDojo",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2607.04434",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "robolab-120",
   "name": "RoboLab-120",
   "full_name": "RoboLab (RoboLab-120)",
   "aliases": [
    "RoboLab-120",
    "NVlabs RoboLab"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Produces a success-rate score for embodied manipulation policies in closed-loop simulation.",
   "summary": {
    "text": "Simulation benchmark of 120 tabletop manipulation tasks for generalist robot policies trained on real-world data.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Authors: Jenai Xuning Yang, Rishit Dagli, Alex Zook, Hugo Hadfield, Ankit Goyal, Stan Birchfield, Fabio Ramos, Jonathan Tremblay. All NVIDIA; Dagli also University of Toronto; Ramos also The University of Sydney.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "builder_type": {
     "value": "platform-vendor",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All authors are NVIDIA employees; NVIDIA also ships the Isaac simulation stack the benchmark runs on. Checked 2026-10-10."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Blog author is at the NVIDIA Seattle Robotics Lab. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2026-04 (arXiv v1 2026-04-10; GitHub repo created 2026-04-08)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "Code tag v0.3.1 published 2026-08-12 (CHANGELOG 0.3.1 dated 2026-08-11: Kinova Gen3 support, floor-standing robots, per-scene ground-height lock). Paper v4 2026-08-14. Latest default-branch commit 2026-09-12 (Cosmos3 policy client update). README news 2026/08: RoboVoLo task library released as an extension.",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Release list at https://github.com/NVlabs/RoboLab/releases; commit date from GitHub API. Checked 2026-10-10."
    },
    "version": {
     "value": "RoboLab-120 (initial task set); code v0.3.1; paper arXiv v4",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "Robotics: Science and Systems XXII (RSS 2026), Sydney (arXiv journal-ref)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "README: task-based evaluation benchmark built on NVIDIA Isaac Lab. Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Tasks: pick-and-place, stacking, rearrangement, tool use with language instructions (README). Competency axes: visual, relational, procedural. Leaderboard reports three levels of language specificity (vague, default, specific). Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Paper experiments: Franka Panda 7-DoF + Robotiq 2F-85, external ZED 2i and wrist ZED mini cameras (DROID-style setup). Repo says tasks are not tied to one robot; v0.3.1 adds Kinova Gen3 (Robotiq 2F-85). Checked 2026-10-10."
    },
    "robots": {
     "value": "Franka Panda + Robotiq 2F-85 (paper); Kinova Gen3 (code v0.3.1)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Blog: 120 human-curated tabletop pick-and-place tasks. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Headline: overall success rate (fully completed episodes); also a normalized graded score with partial credit per subtask. Checked 2026-10-10."
    },
    "metric_detail": {
     "value": "Paper protocol: 10 episodes per task; overall success rate across 120 tasks plus graded score. Leaderboard shows SR% and an undefined 'Score' column; most entries are N = 1,200 episodes (10 per task x 120 tasks, inferred), Lakeside 3,000.",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "evaluator": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "unknown: leaderboard invites submissions via a form; page does not say who runs them",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Official (: 16 entries on 2026-10-10; top Lakeside 51.0% SR (1531/3000), FLUX 3 Action 42.9%, HiDream-O1-Embodied 39.9%; pi0.5 28.0%)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Read the LICENSE file. Checked 2026-10-10."
    },
    "license_data": {
     "value": "No demonstration dataset is distributed (policies fine-tuned on DROID). Bundled 3D object assets carry their own licences: handal, hope, basic, fruits_veggies, objaverse = CC BY-NC-SA 4.0; hot3d = CC BY-SA 4.0 plus HOT3D Dataset License non-sale restriction; ycb = MIT.",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10",
     "display": "unclear: code Apache-2.0 but several bundled asset folders are non-commercial (CC BY-NC-SA 4.0)",
     "note": "Not legal advice. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "correlated",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "display": "Measured (: Spearman rho = 1.00, Pearson r = 0.68 between RoboLab-120 success and RoboArena Elo, 4 policies (pi0.5, pi0-FAST, pi0, PaliGemma). Computed by the RoboLab authors; real numbers come from RoboArena. Authors leave task- and motion-level correlation to future work.)",
     "note": "Authors compared RoboLab-120 success rates with existing RoboArena (real-robot) Elo scores for 4 policies (pi0.5, pi0-FAST, pi0, PaliGemma): Spearman rho = 1.00, Pearson r = 0.68. No new real-robot runs in the paper; correlation computed by the RoboLab authors on RoboArena's numbers. Weak evidence: n = 4 policies, no task-level pairing."
    },
    "citations": {
     "value": 31,
     "display": "31",
     "level": "verified",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar count. Checked 2026-10-10."
    },
    "github_stars": {
     "value": 557,
     "display": "557",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API stargazers_count. Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "protocol-variance",
     "title": "Two supported simulator stacks with different physx builds may give non-comparable results",
     "text": "Checked 2026-10-10.",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies",
     "url": "https://arxiv.org/abs/2604.09860",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-04"
    },
    "s2": {
     "title": "RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies",
     "url": "https://research.nvidia.com/labs/srl/projects/robolab/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "How to Evaluate General-Purpose Robot Policies for Real-World Deployment | NVIDIA Technical Blog",
     "url": "https://developer.nvidia.com/blog/how-to-evaluate-general-purpose-robot-policies-for-real-world-deployment/",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "NVlabs/RoboLab on GitHub (file CHANGELOG.md)",
     "url": "https://github.com/NVlabs/RoboLab/blob/main/CHANGELOG.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "NVlabs/RoboLab on GitHub (repository)",
     "url": "https://github.com/NVlabs/RoboLab",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies (full text)",
     "url": "https://arxiv.org/html/2604.09860v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-04"
    },
    "s7": {
     "title": "RoboLab: A High-Fidelity Simulation Benchmark for Analysis of Task Generalist Policies",
     "url": "https://research.nvidia.com/labs/srl/projects/robolab/leaderboard.html",
     "type": "leaderboard",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "NVlabs/RoboLab on GitHub (blob)",
     "url": "https://github.com/NVlabs/RoboLab/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "NVlabs/RoboLab on GitHub (file THIRD_PARTY_NOTICES.md)",
     "url": "https://github.com/NVlabs/RoboLab/blob/main/THIRD_PARTY_NOTICES.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2604.09860",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "robomimic",
   "name": "robomimic",
   "full_name": "What Matters in Learning from Offline Human Demonstrations for Robot Manipulation",
   "aliases": [
    "RoboMimic",
    "robomimic v0.1 datasets",
    "robomimic (CoRL 2021)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "robomimic is a learning library plus fixed demonstration datasets for five simulated manipulation tasks scored by success rate. Later papers use these tasks and datasets to compare imitation-learning methods (for example Diffusion Policy and Consistency Policy), so it works as a benchmark. The library on its own would be out of scope.",
   "summary": {
    "text": "robomimic is a framework and set of human demonstration datasets from Stanford and UT Austin for teaching robot arms by imitation. Its five simulated tasks, scored by success rate, became a common test for imitation-learning methods, and the easier ones are now near 100%.",
    "short": "robomimic is a set of five simulated robot-arm tasks with human demonstrations, used to test imitation-learning methods (methods that learn by copying demonstrations). The easier tasks are now near 100% success.",
    "sources": [
     "s2",
     "s5",
     "s17"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s5",
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "The authors call robomimic 'a framework for robot learning from demonstration' that lets researchers 'benchmark tasks and algorithms fairly'. Its five simulated tasks have fixed datasets and a success-rate evaluation; Diffusion Policy calls it 'a large-scale robotic manipulation benchmark'. The library alone would not be a benchmark.",
     "short": "Fixed tasks and datasets for imitation learning"
    },
    "kind_secondary": {
     "value": [
      "dataset"
     ],
     "display": "Also human and machine-generated demonstration datasets, and a learning library (the taxonomy has no value for a library)",
     "level": "verified",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "0.5.0",
     "display": "Library v0.5.0 (released 2025-06-27). Datasets: 'robomimic v0.1' (CoRL 2021), now distributed as robosuite v1.5 versions.",
     "level": "verified",
     "sources": [
      "s7",
      "s5",
      "s9"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v0.1.0",
       "display": "Code and paper release dated 2021-08-09 in the README; GitHub release 2021-11-16, 'should be used when trying to reproduce results from the study'",
       "level": "verified",
       "sources": [
        "s5",
        "s7"
       ]
      },
      {
       "value": "v0.2.0 (2021-12-17)",
       "display": "Modular observation modalities; MOMART datasets",
       "level": "verified",
       "sources": [
        "s7"
       ]
      },
      {
       "value": "v0.3.0 (2023-07-04)",
       "display": "BC-Transformer and IQL; robosuite v1.4 and DeepMind MuJoCo bindings",
       "level": "verified",
       "sources": [
        "s7"
       ]
      },
      {
       "value": "v0.4.0 (2025-03-11)",
       "display": "robosuite v1.5 support; datasets moved to Hugging Face",
       "level": "verified",
       "sources": [
        "s7"
       ]
      },
      {
       "value": "v0.5.0 (2025-06-27)",
       "display": "Diffusion Policy, multi-dataset training, language-conditioned policies",
       "level": "verified",
       "sources": [
        "s7"
       ]
      },
      {
       "value": "Three dataset generations",
       "display": "CoRL 2021 datasets used the mujoco-py 'offline_study' branch of robosuite; v0.3 shipped robosuite 1.4.1 versions; v0.4+ ships robosuite 1.5.1 versions. The docs warn learning results 'may not match exactly'.",
       "level": "verified",
       "sources": [
        "s9"
       ]
      }
     ],
     "short": "The library is at 0.5.0 and the datasets are v0.1."
    },
    "publishers": {
     "value": [
      "Stanford University",
      "The University of Texas at Austin"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Ajay Mandlekar, Danfei Xu, Josiah Wong, Soroush Nasiriany, Chen Wang, Rohun Kulkarni, Li Fei-Fei, Silvio Savarese, Yuke Zhu, Roberto Martín-Martín. Part of the ARISE Initiative; development began in the Stanford Vision and Learning Lab in late 2018 (README).",
     "items": [
      {
       "value": "Stanford University",
       "display": "8 of 10 authors (Stanford Vision and Learning Lab)",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ]
      },
      {
       "value": "The University of Texas at Austin",
       "display": "Soroush Nasiriany and Yuke Zhu",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2021-08",
     "display": "arXiv v1 2021-08-06; repository created 2021-08-06; CoRL 2021 (PMLR 164).",
     "level": "verified",
     "sources": [
      "s1",
      "s4",
      "s3",
      "s15"
     ],
     "checked": "2026-10-10",
     "short": "August 2021, at CoRL 2021"
    },
    "latest_update": {
     "value": "2026-08",
     "display": "Last commit 2026-08-09 (lazy CLIP download). Last release v0.5.0 on 2025-06-27. Datasets moved to the robomimic Hugging Face organisation on 2026-02-05.",
     "level": "verified",
     "sources": [
      "s8",
      "s7",
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "August 2026. The latest changes are minor fixes."
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s8",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "No release for over a year; small fixes in 2025-09, 2025-10, 2026-02 and 2026-08.",
     "short": "Maintained. The last release was in June 2025."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Transport needs two arms working together. Tasks are single-goal and use no language instructions in the original study."
    },
    "generalisation": {
     "value": [
      "object-pose"
     ],
     "display": "Object poses are randomized at the start of each episode, within small regions. Objects and scenes stay the same.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "Only the start positions change"
    },
    "venue": {
     "value": "sim",
     "level": "inferred",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Scores in later papers come from the five simulated tasks. The study also ran three real-world versions (Lift, Can, Tool Hang) on a Franka arm; their datasets are released, but real evaluation needs the authors' physical setup."
    },
    "simulator": {
     "value": "robosuite on MuJoCo",
     "display": "robosuite on MuJoCo; v1.5.1 recommended since robomimic v0.4. The CoRL 2021 datasets used robosuite's mujoco-py 'offline_study' branch.",
     "level": "verified",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Policies act at 20 Hz through an operational space controller.",
     "short": "robosuite, built on MuJoCo"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda",
     "display": "Panda arms in simulation and in the real world; Transport uses two.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "tasks": {
     "value": 8,
     "display": "8 tasks: 5 simulated (Lift, Can, Square, Transport, Tool Hang) and 3 real (Lift, Can, Tool Hang)",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Later papers use the 5 simulated tasks, often only 4 (Tool Hang has no multi-human dataset).",
     "short": "5 simulated and 3 real tasks"
    },
    "objects": {
     "value": "fixed objects per task",
     "display": "Lift: a cube. Can: a can and bins. Square: a square nut and a peg. Transport: a hammer, trash cube, bins and lid. Tool Hang: a base frame, hook and wrench.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "1 to 5 fixed objects per task"
    },
    "demonstrations": {
     "value": 3000,
     "display": "3,000 human demonstrations by our arithmetic, plus 5,400 machine-generated trajectories",
     "level": "inferred",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "PH: 200 per task for 5 simulated and 3 real tasks (1,600). MH: 300 per task for 4 tasks (1,200). Paired: 200 (Can). MG: 1,500 (Lift) and 3,900 (Can) from SAC checkpoints.",
     "items": [
      {
       "value": "Proficient-Human (PH)",
       "display": "200 demos per task from one experienced operator via RoboTurk",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      },
      {
       "value": "Multi-Human (MH)",
       "display": "300 demos per task from 6 operators of 3 skill levels (50 each); Lift, Can, Square, Transport",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      },
      {
       "value": "Machine-Generated (MG)",
       "display": "Rollouts from SAC checkpoints: 1,500 (Lift), 3,900 (Can)",
       "level": "verified",
       "sources": [
        "s9"
       ]
      },
      {
       "value": "Paired",
       "display": "200 Can demos: one success and one failure for each of 100 starts",
       "level": "verified",
       "sources": [
        "s2",
        "s9"
       ]
      },
      {
       "value": "Hugging Face size",
       "display": "26 files, about 6.56 GB (low-dim and raw state files; image versions are generated locally)",
       "level": "inferred",
       "sources": [
        "s10"
       ],
       "note": "Summed by us from the file listing (6,557,020,624 bytes)."
      },
      {
       "value": "Real datasets",
       "display": "Lift (1.9 GB), Can (5.3 GB), Tool Hang (58 GB) on a Stanford server",
       "level": "verified",
       "sources": [
        "s9",
        "s12"
       ]
      }
     ],
     "short": "200 to 300 human demonstrations per task"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "metric_detail": {
     "value": "success rate per task and dataset",
     "display": "Success over 50 rollouts, evaluated every few epochs; the study reports the best success over training, averaged over 3 seeds. Results are given per task, per dataset (PH, MH, MG) and per observation type (low-dim or image). There is no standard cross-task average.",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Diffusion Policy reports both the best checkpoint and the average of the last 10 checkpoints.",
     "short": "Success rate per task, taken from the best checkpoint saved during training"
    },
    "trials": {
     "value": "50 rollouts per evaluation",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Study",
       "display": "50 rollouts every 50 epochs (low-dim) or 20 epochs (image); 3 seeds; best over training",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Study, real robot",
       "display": "Final checkpoint, 30 rollouts",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Diffusion Policy",
       "display": "3 seeds x 50 initial conditions (150); best checkpoint and average of last 10 checkpoints",
       "level": "verified",
       "sources": [
        "s17"
       ]
      }
     ],
     "short": "50 attempts per evaluation, averaged over 3 training runs (seeds)"
    },
    "uncertainty_reported": {
     "value": "sometimes",
     "level": "inferred",
     "sources": [
      "s2",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "The study reports mean and standard deviation over 3 seeds. Diffusion Policy gives seed averages without error bars."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s5",
      "s14"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s5",
      "s14",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the site, docs or README."
    },
    "top_score": {
     "value": "per task",
     "display": "No single headline score. Since Diffusion Policy (2023), the best-checkpoint success is 1.00 on most task variants; Tool Hang and multi-human Transport still show gaps.",
     "level": "inferred",
     "sources": [
      "s17",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Not charted: papers report different subsets and checkpoint rules. Values below are success rates as each paper prints them.",
     "items": [
      {
       "value": "Tool Hang PH, image",
       "display": "Diffusion Policy-C 0.95 best / 0.73 last-10 average (2023); study BC-RNN 67.3 (2021)",
       "level": "verified",
       "sources": [
        "s17",
        "s2"
       ]
      },
      {
       "value": "Transport MH, image",
       "display": "Diffusion Policy-C 0.89 / 0.69 (2023); study BC-RNN 42.0 (2021)",
       "level": "verified",
       "sources": [
        "s17",
        "s2"
       ]
      },
      {
       "value": "Square MH, image",
       "display": "Diffusion Policy-C 0.98 / 0.84 (2023); study BC-RNN 76.7 (2021)",
       "level": "verified",
       "sources": [
        "s17",
        "s2"
       ]
      },
      {
       "value": "Lift, Can (PH and MH), image",
       "display": "Diffusion Policy-C 1.00 best on all four (2023)",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "Tool Hang PH, state",
       "display": "Diffusion Policy-T 1.00 / 0.87; Diffusion Policy-C 0.50 / 0.30 (2023)",
       "level": "verified",
       "sources": [
        "s17"
       ]
      }
     ],
     "short": "Reported per task. Most task variants are at 1.00 (100%)."
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, copyright 2021 Stanford Vision and Learning Lab."
    },
    "license_data": {
     "value": "MIT",
     "display": "Hugging Face dataset card: mit (simulated datasets)",
     "level": "verified",
     "sources": [
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "The card says the repository holds 'some of the datasets'. The real-robot datasets on the Stanford server carry no licence statement that we found."
    },
    "license_assets": {
     "value": "MIT",
     "display": "Task assets ship in robosuite, whose LICENSE is MIT (with an Apache-2.0 notice for partial MuJoCo code). No separate asset licence.",
     "level": "inferred",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Reading the robosuite root licence as covering its bundled meshes is ours; robosuite has no other licence file.",
     "short": "MIT, through robosuite, by our reading"
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s10",
      "s12",
      "s25",
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Simulated datasets on an ungated Hugging Face repository (moved to robomimic/robomimic_datasets on 2026-02-05; the old amandlek/robomimic id redirects). Real datasets on a Stanford server: Lift and Can links answered HTTP 200 on 2026-10-10; the 58 GB Tool Hang link returned HTTP 503 twice.",
     "short": "Open. The data is on Hugging Face."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s6",
      "s10",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Code, simulated datasets and robosuite are MIT. The real-robot datasets have no stated licence, so their use is unclear. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "The study trained BC-RNN on three real-world versions of its tasks with settings tuned in simulation. No paired comparison of the same policies in simulation and on real robots.",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Real tasks were built to match the simulated ones in look, size and start randomization. BC-RNN on 200 real demos per task, final checkpoint, 30 rollouts: Lift 96.7%, Can 73.3%, Tool Hang 3.3%. Removing image randomization or the wrist camera hurt in both simulation and the real Can task (26.7% and 43.3%), which the authors read as 'study results transfer to real-world settings'. Real policies are different policies trained on real data, so this is not a score comparison.",
     "short": "The authors ran real-robot versions of three tasks. No study compares the same policies' scores in simulation and on real robots."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Benchmark scores come from simulation."
    },
    "citations": {
     "value": 1112,
     "display": "1,112 (Semantic Scholar; 131 influential)",
     "level": "verified",
     "sources": [
      "s24"
     ],
     "checked": "2026-10-10",
     "short": "1,112"
    },
    "github_stars": {
     "value": 1578,
     "display": "1,578 stars, 430 forks (ARISE-Initiative/robomimic)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "1,578"
    },
    "dataset_downloads": {
     "value": 4677,
     "display": "4,677 (Hub 'downloads' field), 38,114 all time, 4 likes: robomimic/robomimic_datasets",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "We did not check the time window behind the 'downloads' field.",
     "short": "4,677 on Hugging Face"
    },
    "used_by": {
     "value": "No count of papers reporting robomimic results was found. Named users below.",
     "level": "verified",
     "sources": [
      "s17",
      "s18",
      "s19",
      "s20",
      "s21",
      "s22",
      "s23"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Diffusion Policy",
       "display": "2023-03: all five tasks, PH and MH, state and image",
       "level": "verified",
       "sources": [
        "s17"
       ]
      },
      {
       "value": "AWE",
       "display": "Stanford, 2023-07: Lift, Can, Square in low-data settings",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "MimicGen",
       "display": "NVIDIA and UT Austin, 2023-10: uses robomimic Square demos as source data and robomimic for training",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Consistency Policy",
       "display": "2024-05: single-robot robomimic tasks",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "DPPO",
       "display": "2024-09: RL fine-tuning on Lift, Can, Square, Transport",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "RoboCasa, LIBERO",
       "display": "Use robomimic code: RoboCasa's policy learning is a robomimic branch; LIBERO depends on robomimic 0.2.0",
       "level": "verified",
       "sources": [
        "s22",
        "s23"
       ]
      }
     ],
     "short": "A common test for imitation-learning methods"
    },
    "industry_use": {
     "value": [
      "NVIDIA"
     ],
     "level": "verified",
     "sources": [
      "s21",
      "s22"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "NVIDIA",
       "display": "MimicGen (NVIDIA and UT Austin) builds on robomimic data and code; RoboCasa (UT Austin and NVIDIA) trains with a robomimic branch",
       "level": "verified",
       "sources": [
        "s21",
        "s22"
       ]
      }
     ]
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "saturated",
     "title": "Top methods score 100% on most task variants",
     "text": "Diffusion Policy (2023) reports best-checkpoint success of 1.00 on Lift and Can (PH and MH) and on Square PH and Transport PH with image inputs. AWE (2023) says Diffusion Policy is near-perfect on Lift, Can and Square with 200 proficient demonstrations, so it studies low-data settings instead. DPPO reports RL fine-tuning converging to about 100% on Lift and Can.",
     "short": "Since 2023, top methods have reported a success rate of 1.00 (100%) on most task variants.",
     "level": "verified",
     "sources": [
      "s17",
      "s18",
      "s20"
     ],
     "status": "open"
    },
    {
     "id": "i2",
     "type": "protocol-variance",
     "title": "Scores depend on which checkpoint is reported",
     "text": "The study evaluates every checkpoint in the test environment and reports the best, with no separate validation set. Its own analysis finds that picking by validation loss or taking the last checkpoint gives 10% to 100% lower success. Diffusion Policy prints both numbers: for example 0.68 best vs 0.46 last-10 average on Transport MH (state, CNN) and 0.76 vs 0.47 on Tool Hang PH (image, Transformer).",
     "short": "Scores from the best checkpoint (a saved copy of the model during training) can be far above scores from the last checkpoint. The robomimic study found 10% to 100% lower success when it took the last checkpoint or picked one by validation loss.",
     "level": "verified",
     "sources": [
      "s2",
      "s17"
     ],
     "status": "open"
    },
    {
     "id": "i3",
     "type": "inconsistent-reporting",
     "title": "Dataset versions and subsets differ between papers",
     "text": "Three dataset generations exist (mujoco-py offline_study, robosuite 1.4.1, robosuite 1.5.1), and the docs warn results may not match the CoRL 2021 datasets. Image observations are re-rendered locally from raw states. Papers choose PH or MH data, state or image inputs, and four or five tasks. Diffusion Policy's re-run of BC-RNN scored slightly better than the original paper.",
     "short": "Papers use different dataset versions, subsets and inputs. Their results are rarely directly comparable.",
     "level": "verified",
     "sources": [
      "s9",
      "s17",
      "s20"
     ],
     "status": "open"
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "On the proficient-human (PH) data, Lift, Can and Square no longer separate strong methods. Differences now show on Tool Hang, on Transport with mixed-quality data, and in low-data settings.",
     "short": "The easy tasks no longer separate methods. Look at results on Tool Hang and on the multi-human (MH) data.",
     "basis": [
      "issues.i1",
      "facts.top_score"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r2",
     "text": "Compare robomimic numbers only when the dataset type, observation type, dataset version and checkpoint rule all match. Best-checkpoint numbers are optimistic because the checkpoint is chosen on the test environment.",
     "short": "Compare numbers only when the dataset version and the checkpoint rule match.",
     "basis": [
      "issues.i2",
      "issues.i3",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10"
    },
    {
     "id": "r3",
     "text": "A robomimic score says little about language, new objects, new scenes or real robots. It is useful as a small, reproducible test of imitation-learning algorithms, which is how most later papers use it.",
     "short": "robomimic is a small test of learning algorithms. It does not test general robot skill.",
     "basis": [
      "facts.capability",
      "facts.generalisation",
      "facts.sim_to_real",
      "facts.used_by"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10"
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy will do on a real robot.",
     "sub": "Only the authors have run the real-robot versions of the tasks. No study compares the same policies in simulation and on real robots.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Which of the top methods is better.",
     "sub": "Top methods reach 100% on most task variants, so these variants no longer separate them.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "How well a policy handles new objects, scenes or instructions.",
     "sub": "Each task uses the same objects every time, and the tasks have no language instructions.",
     "basis": [
      "facts.generalisation",
      "facts.capability"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "Study paper full text (Section 4.7, Appendix E.2); study page; docs; Diffusion Policy, DPPO, MimicGen (real experiments use other tasks); evidence papers PolaRiS (2512.16881), SureSim (2510.04354), 'A Practical Recipe Towards Improving Sim-and-Real Correlation' (2606.10366), Betting for Sim-to-Real (2604.24018), Robot Policy Evaluation for Sim-to-Real Transfer (2508.11117), Active Real-World Factor-Based Evaluation (2607.14439), Beyond Binary Success (2603.13616): no robomimic mention in full text; web searches. No paired study found.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data (real datasets)",
     "where": "Docs datasets page, README, Hugging Face card, dataset registry: no licence statement for the Stanford-hosted real datasets.",
     "date": "2026-10-10"
    },
    {
     "for": "used_by (count)",
     "where": "No tracker found; the docs' 'Projects using robomimic' page only links to Google Scholar.",
     "date": "2026-10-10"
    },
    {
     "for": "issues (audits)",
     "where": "The 2026 audit (2606.04233) does not cover robomimic; web search for robomimic saturation found AWE's statement only.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "What Matters in Learning from Offline Human Demonstrations for Robot Manipulation (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2108.03298",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2021-08",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "robomimic study paper, full text v2 (Tables 1 to 3, Sections 3 and 4, Appendix E)",
     "url": "https://arxiv.org/pdf/2108.03298v2",
     "type": "paper",
     "publisher": "arXiv (CoRL 2021 text)",
     "date": "2021-09",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "PMLR volume 164 page (Proceedings of the 5th Conference on Robot Learning, 164:1678-1690)",
     "url": "https://proceedings.mlr.press/v164/mandlekar22a.html",
     "type": "paper",
     "publisher": "CoRL 2021 / PMLR",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "GitHub API: ARISE-Initiative/robomimic (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/ARISE-Initiative/robomimic",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "robomimic README (latest updates, description)",
     "url": "https://github.com/ARISE-Initiative/robomimic/blob/master/README.md",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "robomimic LICENSE (MIT, 2021 Stanford Vision and Learning Lab)",
     "url": "https://github.com/ARISE-Initiative/robomimic/blob/master/LICENSE",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "robomimic GitHub releases (v0.1.0 to v0.5.0) and release notes",
     "url": "https://github.com/ARISE-Initiative/robomimic/releases",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2025-06-27",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "robomimic commit history (master)",
     "url": "https://github.com/ARISE-Initiative/robomimic/commits/master",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2026-08-09",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "robomimic docs: robomimic v0.1 (CoRL 2021) datasets, versions warning, dataset info, reproduction steps",
     "url": "https://robomimic.github.io/docs/datasets/robomimic_v0.1.html",
     "type": "site",
     "publisher": "ARISE Initiative",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "Hugging Face dataset robomimic/robomimic_datasets: card (license mit) and file listing",
     "url": "https://huggingface.co/datasets/robomimic/robomimic_datasets",
     "type": "repo",
     "publisher": "robomimic team",
     "date": "2025-04-13",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Hugging Face Hub API record for robomimic/robomimic_datasets (downloads, likes)",
     "url": "https://huggingface.co/api/datasets/robomimic/robomimic_datasets?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "robomimic dataset registry (robomimic/__init__.py: horizons, real-data links on Stanford server)",
     "url": "https://github.com/ARISE-Initiative/robomimic/blob/master/robomimic/__init__.py",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2026-02-05",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "robomimic pull request #294 'update HF_REPO_ID' (datasets moved to a new Hugging Face organisation)",
     "url": "https://github.com/ARISE-Initiative/robomimic/pull/294",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2026-02-05",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "robomimic project site",
     "url": "https://robomimic.github.io/",
     "type": "site",
     "publisher": "Stanford University and UT Austin",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "robomimic study page",
     "url": "https://robomimic.github.io/study/",
     "type": "site",
     "publisher": "robomimic team",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "robosuite LICENSE (MIT, 2022 Stanford Vision and Learning Lab and UT Robot Perception and Learning Lab)",
     "url": "https://github.com/ARISE-Initiative/robosuite/blob/master/LICENSE",
     "type": "repo",
     "publisher": "ARISE Initiative",
     "date": "2022",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Diffusion Policy: Visuomotor Policy Learning via Action Diffusion (Tables 1 and 2)",
     "url": "https://arxiv.org/abs/2303.04137",
     "type": "paper",
     "publisher": "arXiv (RSS 2023; IJRR 2024)",
     "date": "2023-03",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "Waypoint-Based Imitation Learning for Robotic Manipulation (AWE)",
     "url": "https://arxiv.org/abs/2307.14326",
     "type": "paper",
     "publisher": "arXiv (Stanford)",
     "date": "2023-07",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Consistency Policy: Accelerated Visuomotor Policies via Consistency Distillation",
     "url": "https://arxiv.org/abs/2405.07503",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Diffusion Policy Policy Optimization (DPPO)",
     "url": "https://arxiv.org/abs/2409.00588",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-09",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "MimicGen: A Data Generation System for Scalable Robot Learning using Human Demonstrations",
     "url": "https://arxiv.org/abs/2310.17596",
     "type": "paper",
     "publisher": "arXiv (NVIDIA, UT Austin; CoRL 2023)",
     "date": "2023-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "RoboCasa v0.2 docs: Policy Learning (official code is a robomimic branch)",
     "url": "https://github.com/robocasa/robocasa/blob/v0.2/docs/use_cases/policy_learning.md",
     "type": "repo",
     "publisher": "RoboCasa team",
     "date": "2025-04",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "LIBERO requirements.txt (robomimic==0.2.0)",
     "url": "https://github.com/Lifelong-Robot-Learning/LIBERO/blob/master/requirements.txt",
     "type": "repo",
     "publisher": "Lifelong-Robot-Learning",
     "date": "2024-12-10",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Semantic Scholar API record for arXiv:2108.03298",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2108.03298?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "robomimic real-robot datasets on the Stanford download server (Lift, Can, Tool Hang)",
     "url": "http://downloads.cs.stanford.edu/downloads/rt_benchmark/",
     "type": "repo",
     "publisher": "Stanford University",
     "date": "2021",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created full entry from primary sources (no basic entry existed). Scope decided: in scope as a benchmark of fixed tasks and datasets."
    }
   ]
  },
  {
   "id": "robomind",
   "name": "RoboMIND",
   "full_name": "RoboMIND: Benchmark on Multi-embodiment Intelligence Normative Data for Robot Manipulation",
   "aliases": [
    "RoboMIND 1.0",
    "RoboMIND v1.0",
    "RoboMIND v1.1",
    "RoboMIND v1.2",
    "Multi-embodiment Intelligence Normative Data for Robot Manipulation"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "A teleoperated training dataset whose paper reports one-off real-robot tests and a small check of its digital twin against real results. There is no public evaluation protocol, test server or leaderboard for version 1.x. The Isaac Sim evaluation code that X-Humanoid released (RoboMIND-Sim) belongs to RoboMIND 2.0, a separate record.",
   "summary": {
    "text": "RoboMIND is a robot-manipulation dataset from X-Humanoid (Beijing Innovation Center of Humanoid Robotics) with Peking University and BAAI: about 107,000 teleoperated trajectories on four robot types and 479 tasks, of which about 30,000 were recorded in a simulated digital twin. Its paper tests policies on real robots and checks the twin against real results on five tasks.",
    "sources": [
     "s2",
     "s7"
    ],
    "short": "RoboMIND is a dataset of about 107,000 demonstrations on four robot types, used to train robot control models (policies). About 28% of the demonstrations come from simulation, and its paper tests policies on real robots."
   },
   "facts": {
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The title says 'Benchmark', and the paper says RoboMIND 'serves as a benchmark' for the methods it tests. What is released is training data plus conversion and training code; the real-robot test setups are not released."
    },
    "kind_secondary": {
     "value": [
      "study"
     ],
     "display": "Its scores come from one-off tests run by the builders",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All success rates in the paper come from the authors' own real-robot tests and their own digital twin."
    },
    "version": {
     "value": "v1.2",
     "display": "Current release v1.2 (dataset card). Paper: arXiv v1 2024-12-18, v2 2025-02-14, v3 2025-05-27 (RSS 2025 version). Release folders benchmark1_0, benchmark1_1, benchmark1_2.",
     "level": "verified",
     "sources": [
      "s7",
      "s1",
      "s13"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "v1.0",
       "display": "arXiv v1 (2024-12): 55k trajectories, 279 tasks, 61 object classes, 36 skills, 308.6 hours, 4 robot types. The card's Version 1.0 note says 69 object classes.",
       "level": "verified",
       "sources": [
        "s3",
        "s7"
       ]
      },
      {
       "value": "v1.1",
       "display": "107k trajectories, 479 tasks, 96 object classes (card). Matches arXiv v2 (2025-02) and v3.",
       "level": "verified",
       "sources": [
        "s7",
        "s4",
        "s2"
       ]
      },
      {
       "value": "v1.2",
       "display": "Adds 10 'Upright_Cup' tasks: 1 real-world task and 9 from the digital twin, varying mug placement, table texture and mug appearance. Release date not stated.",
       "level": "verified",
       "sources": [
        "s7",
        "s13"
       ]
      },
      {
       "value": "RoboMIND 2.0 (separate dataset)",
       "display": "Same lead organisation, arXiv 2025-12-31: over 310K dual-arm trajectories on six robot types, mobile and tactile data, and an Isaac Sim benchmark (RoboMIND-Sim). Its paper calls the earlier set 'RoboMIND 1.0' and says 1.0 focuses on single-arm manipulation. The v1 card and site announce it as 'RoboMIND V2.0'. Separate Atlas record (robomind-2-0).",
       "level": "verified",
       "sources": [
        "s22",
        "s7",
        "s18",
        "s23"
       ]
      }
     ],
     "short": "v1.2. RoboMIND 2.0 is a separate dataset."
    },
    "publishers": {
     "value": [
      "Beijing Innovation Center of Humanoid Robotics (X-Humanoid)",
      "Peking University",
      "Beijing Academy of Artificial Intelligence"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s5",
      "s18"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Beijing Innovation Center of Humanoid Robotics (X-Humanoid)",
       "display": "北京人形机器人创新中心有限公司. Builds the Tien Kung humanoid. Corresponding author Jian Tang; project leaders Zhengping Che and Xiaozhu Ju.",
       "level": "verified",
       "sources": [
        "s2",
        "s20"
       ]
      },
      {
       "value": "Peking University",
       "display": "State Key Laboratory of Multimedia Information Processing, School of Computer Science. Corresponding author Shanghang Zhang.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Beijing Academy of Artificial Intelligence",
       "display": "Second affiliation of several PKU authors.",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ]
      }
     ],
     "note": "37 authors on the RSS page.",
     "short": "X-Humanoid, with Peking University and BAAI"
    },
    "builder_type": {
     "value": "robot-company",
     "level": "verified",
     "sources": [
      "s20",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The lead organisation's official site is titled '北京人形机器人创新中心有限公司' (a limited company). It builds the Tien Kung humanoid used in the data."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All affiliations are in Beijing."
    },
    "first_release": {
     "value": "2024-12",
     "display": "arXiv v1 on 2024-12-18; Hugging Face dataset created 2025-01-02",
     "level": "verified",
     "sources": [
      "s1",
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "December 2024"
    },
    "latest_update": {
     "value": "2026-04",
     "display": "Hugging Face files last changed 2026-04-14, the day the data-check repository (RoboMIND-dataset-utils) was created. ModelScope copy last updated 2026-07-20. Training toolchain last pushed 2026-09-16.",
     "level": "verified",
     "sources": [
      "s8",
      "s16",
      "s12",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "The data last changed in April 2026."
    },
    "published_at": {
     "value": "RSS 2025",
     "display": "Robotics: Science and Systems XXI (Los Angeles, June 21-25, 2025), paper 152, DOI 10.15607/RSS.2025.XXI.152",
     "level": "verified",
     "sources": [
      "s5",
      "s6",
      "s21"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "dexterous",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Six task categories (articulated, coordination, basic, multi-object, precision, scene understanding). Over 75% of Tien Kung and AgileX tasks combine two or more skills. Every task has a language description; 10k trajectories have frame-level language annotations."
    },
    "generalisation": {
     "value": [
      "object-instance",
      "visual"
     ],
     "display": "Most tests reuse the training tasks. One task was also tested with new objects and new tablecloths.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Table VI: FR-PlaceBreadPlate with the bread swapped for corn, banana or apple, and with three unseen tablecloths.",
     "short": "Only one task is tested with new objects and backgrounds."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Headline results come from real robots. The paper also evaluates policies in its Isaac Sim digital twin on 6 tasks (see sim_to_real and validity). The v1 twin environment is not released; only its recorded trajectories are."
    },
    "simulator": {
     "value": "NVIDIA Isaac Sim (digital twin; version not stated)",
     "level": "verified",
     "sources": [
      "s2",
      "s7",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Used for the twin data and the twin evaluations. In the released simulation data, robot state is sampled at about four times the camera rate, and depth images are not yet available (card). The Isaac Sim code released later (RoboMIND-Sim, Isaac Sim 4.5 and 5.1) is for RoboMIND 2.0 Tien Kung tasks.",
     "short": "Isaac Sim, used only for the digital twin (a simulated copy of the real set-up)"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm",
      "humanoid",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "robots": {
     "value": "Franka Emika Panda, UR5e, AgileX Cobot Magic V2.0, Tien Kung humanoid",
     "display": "Franka Emika Panda (3 RealSense D435i cameras, Robotiq gripper); UR5e (top RealSense camera, Robotiq gripper); AgileX Cobot Magic V2.0 dual-arm (3 Orbbec cameras); X-Humanoid Tien Kung humanoid (42 DoF, two Inspire RH56DFX dexterous hands, Orbbec Gemini 335 cameras)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The v1.1 release also has folders for a dual-arm Franka FR3 setup and a second Tien Kung variant (h5_franka_fr3_dual, h5_tienkung_prod1_gello_1rgb), which the paper does not describe.",
     "short": "4 robot types"
    },
    "scene": {
     "value": [
      "kitchen",
      "home",
      "office-lab",
      "retail-logistics",
      "industrial"
     ],
     "display": "Lab setups themed as kitchen 43.4%, domestic 26.7%, office 16.5%, retail 6.9%, industrial 6.5% of trajectories",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Shares read from Figure 1(d) of the RSS version. The authors list 'relatively simple background environments' as a limitation.",
     "short": "Lab tables themed as kitchen, home or office"
    },
    "tasks": {
     "value": 479,
     "display": "479 tasks (v1.1 and v1.2); 279 in v1.0",
     "level": "verified",
     "sources": [
      "s2",
      "s7",
      "s3",
      "s15"
     ],
     "checked": "2026-10-10",
     "note": "Tasks are defined by robot, skill, objects and scene. The v1.2 instruction file lists 479 rows with 468 distinct task names (counted by us).",
     "short": "479 tasks"
    },
    "objects": {
     "value": 96,
     "display": "96 object classes (v1.1 and v1.2); 61 (arXiv v1) or 69 (card) in v1.0",
     "level": "verified",
     "sources": [
      "s2",
      "s3",
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "96 object classes"
    },
    "demonstrations": {
     "value": "107k",
     "display": "107k trajectories (305.5 hours) on four robot types. By the paper's own breakdown, 30,035 of them (about 28%) come from the simulated digital twin, although the site and card call all 107k 'real-world'.",
     "level": "verified",
     "sources": [
      "s2",
      "s7",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Per-robot counts differ between the paper text and its own Figure 1 / the card; resolved as far as possible in the items. See issues.i1 and issues.i2.",
     "items": [
      {
       "value": "Paper text: real per robot plus a simulation total",
       "display": "Franka 26,856; Tien Kung 15,187; AgileX 10,269; UR5e 25,170; simulation 30,035. Sum 107,517 (our arithmetic).",
       "level": "verified",
       "sources": [
        "s2",
        "s6"
       ]
      },
      {
       "value": "Figure 1(a) and dataset card: totals per robot including simulation",
       "display": "Franka 52,926; humanoid (Tien Kung) 19,152; AgileX 10,629; UR5e 25,170. Sum 107,877 (our arithmetic). The card calls all of them teleoperation data.",
       "level": "verified",
       "sources": [
        "s6",
        "s7"
       ]
      },
      {
       "value": "How the two fit",
       "display": "Franka 52,926 = 26,856 real + 26,070 twin trajectories (Section IV-A gives 'over 26,070' twin and 26,866 real). Tien Kung 19,152 = 15,187 real + 3,965 twin, by our arithmetic (30,035 − 26,070); the paper's 17.8% humanoid share and its '19k' humanoid ablation fit this. The v1.1 release has a Tien Kung simulation folder, although Section IV-A says the other three robots have real data only. AgileX 10,629 (figure, card) vs 10,269 (text) is unresolved; the 360 difference is the whole gap between the two totals.",
       "level": "inferred",
       "sources": [
        "s2",
        "s6",
        "s7",
        "s13",
        "s14"
       ],
       "note": "Release archives are packed as split tar.gz files, so per-folder counts could not be checked without downloading terabytes."
      },
      {
       "value": "v1.0 breakdown",
       "display": "Franka 19,222; Tien Kung 9,686; AgileX 8,030; UR-5e 6,911; simulation 11,783 (all Franka). Sum 55,632 (our arithmetic). 308.6 hours.",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "Average length",
       "display": "Franka 179 frames, UR 158, AgileX 655, humanoid 669 (Figure 1(b))",
       "level": "verified",
       "sources": [
        "s6"
       ]
      },
      {
       "value": "Failures and annotations",
       "display": "5k failure demonstrations with causes; 10k trajectories with frame-level language annotations (Gemini drafts revised by people)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "12.3 TB",
       "display": "Total file size on Hugging Face",
       "level": "verified",
       "sources": [
        "s7"
       ]
      }
     ],
     "short": "107k trajectories, about 28% of them simulated"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Testers record success or failure, and the cause of each failure from 9 predefined categories."
    },
    "metric_detail": {
     "value": "success rate over 10 real trials per task and model",
     "display": "Each trained model is run 10 times per task on the real robot and the share of successes is reported, with failure causes logged.",
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Single-task imitation learning",
       "display": "ACT, Diffusion Policy and BAKU trained from scratch on 45 tasks (Franka 15, Tien Kung 10, AgileX 15, UR5e 5). ACT averages: AgileX 55.3%, UR5e 38.0%, Tien Kung 34.0%, Franka 30.7%.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Vision-language-action models",
       "display": "OpenVLA (Franka only), RDT-1B and CrossFormer fine-tuned per robot on 15 tasks. Example: RDT-1B 10/10 on AX-AppleYellowPlate; CrossFormer 0/10 on all five AgileX tasks.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Pretraining on all of RoboMIND",
       "display": "RDT-1B and CrossFormer pretrained on the full set, then fine-tuned on about 1% of it, improved on most of the 15 tasks (Table IV). Excluding the humanoid data (19k of 107k) lowered RDT-1B's average on 5 Franka tasks from 0.68 to 0.6.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Most common failure",
       "display": "For ACT, 'Inaccurate Positioning' is the top failure cause on every robot type (48% of failures on the humanoid).",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "short": "Success rate over 10 real-robot trials per task"
    },
    "trials": {
     "value": 10,
     "display": "10 per task and model",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "10 per task"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Results are given as successes out of 10 or as percentages, with no error bars or intervals."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All results come from the authors' own tests."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s18",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard, test server or challenge on the project site, dataset card or website repo."
    },
    "license_code": {
     "value": "Apache-2.0",
     "display": "Apache-2.0 for the training toolchain the paper links (x-humanoid-training-toolchain) and for the data-check scripts (RoboMIND-dataset-utils)",
     "level": "verified",
     "sources": [
      "s17",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Toolchain LICENSE: Apache 2.0, 'Copyright 2025 The Beijing Innovation Center of Humanoid Robotics'. The project-website repo has an Apache 2.0 LICENSE file, which GitHub reports as NOASSERTION."
    },
    "license_data": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s8",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Hugging Face card metadata 'apache-2.0'; ModelScope copy 'apache-2.0'."
    },
    "license_assets": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "The digital-twin scenes and 3D assets for version 1 are not released, and no licence for them is stated. Looked at the paper, dataset card, project site, ModelScope file tree and the X-Humanoid GitHub organisation."
    },
    "access": {
     "value": "registration",
     "display": "Hugging Face gate with automatic approval (agree to share contact information). Also on ModelScope and BAAI's data platform.",
     "level": "verified",
     "sources": [
      "s7",
      "s8",
      "s12",
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Hub API: gated = 'auto'. The ModelScope file tree is publicly listable; its download conditions (ApprovalMode 1, ProtectedMode 2) were not tested.",
     "short": "Free. Users agree to share contact details and are approved automatically."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s8",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Data and code are Apache-2.0, which allows commercial use. No third-party assets are distributed. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Scores come from real robots. The authors also checked their digital twin: Pearson r 0.83 and 0.91 over 5 tasks for 2 policies; the twin scored far below the real robot.",
     "level": "inferred",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; the numbers are verified (Table VII). The correlation is across 5 tasks for one policy at a time, with 10 trials per task, so each r rests on 5 points and no interval is given. We recomputed both values from the table. Average successes were ACT 1.4/10 in the twin vs 5.4/10 real and Diffusion Policy 3.2/10 vs 8.0/10. On all 5 tasks the twin ordered ACT and Diffusion Policy the same way as the real robot (our reading of Table VII). In the co-training study (Figure 17) a policy trained only on twin data scored 0.9 in the twin and 0.1 on the real robot. The v1 twin is not released, so no one else can repeat the check. No independent study found.",
     "short": "The authors checked their digital twin against real robots. Correlation r = 0.83 and 0.91."
    },
    "real_reproducibility": {
     "value": "none",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All real-robot tests ran on X-Humanoid's own setups; no other site has reported running them."
    },
    "citations": {
     "value": 244,
     "display": "244 (Semantic Scholar; 13 influential)",
     "level": "verified",
     "sources": [
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Read after several rate-limited attempts.",
     "short": "244"
    },
    "github_stars": {
     "value": 139,
     "display": "139 stars on the project-website repo; 79 on the training toolchain",
     "level": "verified",
     "sources": [
      "s19",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "139"
    },
    "dataset_downloads": {
     "value": 30259,
     "display": "Hugging Face: 30,259 (Hub 'downloads' field), 419,215 all time, 54 likes. ModelScope: 485,336.",
     "level": "verified",
     "sources": [
      "s8",
      "s12"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "'Over 20 million' (reported)",
       "display": "Beijing News (2026-09-02) quotes the centre: RoboMIND downloads doubled in a month to over 20 million. Its other figures (300,000+ dual-arm trajectories, 700+ tasks, open since last December) describe RoboMIND 2.0, so the claim is not about version 1 alone.",
       "level": "reported",
       "sources": [
        "s26"
       ]
      }
     ],
     "short": "30,259 on Hugging Face"
    },
    "used_by": {
     "value": "Used as pretraining data by EO-1 and X-Humanoid's XR-1. Its paper tests ACT, Diffusion Policy, BAKU, OpenVLA, RDT-1B and CrossFormer.",
     "level": "verified",
     "sources": [
      "s24",
      "s25",
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list. No outside paper was found that reports scores on RoboMIND tasks.",
     "items": [
      {
       "value": "EO-1",
       "display": "2025-08. Trained on AgiBotWorld, Open X-Embodiment, RoboMIND and other data. Hugging Face lists EO-1-3B as trained on RoboMIND.",
       "level": "verified",
       "sources": [
        "s24",
        "s7"
       ]
      },
      {
       "value": "XR-1",
       "display": "X-Humanoid, 2025-11. Pretraining mixes Open-X, RoboMIND, Ego4D and XR-D (a subset of RoboMIND 2.0).",
       "level": "verified",
       "sources": [
        "s25"
       ]
      }
     ],
     "short": "Used as pretraining data for EO-1 and XR-1"
    },
    "industry_use": {
     "value": [
      "X-Humanoid"
     ],
     "level": "verified",
     "sources": [
      "s17",
      "s25"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "X-Humanoid",
       "display": "Ships a LeRobot-based toolchain for training Tien Kung robots on RoboMIND, and pretrains its XR-1 model on it.",
       "level": "verified",
       "sources": [
        "s17",
        "s25"
       ]
      }
     ]
    },
    "status": {
     "value": "maintained",
     "display": "Data checks and fixes in 2026-04; new collection goes into RoboMIND 2.0",
     "level": "inferred",
     "sources": [
      "s8",
      "s16",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "Five open user discussions on the Hugging Face card (2025-12 to 2026-08) have no maintainer reply visible.",
     "short": "Maintained. New data collection goes into RoboMIND 2.0."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Trajectory counts differ between versions, the paper's figure and its text",
     "text": "Version 1.0 (arXiv v1) has 55k trajectories and 308.6 hours; versions 1.1 and 1.2 have 107k trajectories but 305.5 hours. Within the RSS paper, the text lists Franka 26,856, Tien Kung 15,187, AgileX 10,269, UR5e 25,170 and simulation 30,035, while Figure 1(a) and the card list Franka 52,926, humanoid 19,152, AgileX 10,629 and UR5e 25,170. Most of the gap is explained by Figure 1 and the card folding twin data into each robot (see facts.demonstrations). Still unresolved: AgileX 10,629 vs 10,269; Franka real 26,856 (introduction) vs 26,866 (Section IV-A); object classes 61 vs 69 for v1.0; and the drop in hours from 308.6 to 305.5 while trajectories doubled.",
     "level": "verified",
     "sources": [
      "s3",
      "s2",
      "s6",
      "s7"
     ],
     "status": "open",
     "short": "The number of trajectories differs between versions, between the paper's text and its figure, and on the dataset card. Some of these differences are still unexplained."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "About 28% of the 'real-world' data is simulated",
     "text": "The project site and dataset card describe '107k real-world demonstration trajectories'. The paper's own breakdown counts 30,035 of them from the Isaac Sim digital twin, and Franka's 49.2% share includes over 26,070 twin trajectories. The paper's comparison table places RoboMIND among real-world datasets. Version 1.0 also mixed in 11,783 simulated trajectories while its abstract said '55k real-world'.",
     "level": "verified",
     "sources": [
      "s18",
      "s7",
      "s2",
      "s3"
     ],
     "status": "open",
     "short": "The site and dataset card describe '107k real-world' trajectories. By the paper's own breakdown, about 30,000 of them come from simulation."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "End-effector data is frozen in some subsets",
     "text": "X-Humanoid's data-check repository (2026-04) reports end-effector values that stay almost constant in three subsets: h5_ur_1rgb, h5_simulation and h5_sim_franka_3rgb, probably from a blocking problem during collection. It recommends using joint positions and recomputing end-effector poses with forward kinematics. Users report constant UR end-effector poses (discussion #8) and UR episodes whose joints barely move although the video shows motion (discussion #6); one user posted a scripted count of 22,987 of 25,721 UR episodes (89.37%) with the first six joint values unchanged, which X-Humanoid has not answered. The card also notes BGR image order in four subsets and RGB in the rest, and 675 Franka trajectories with only two of three cameras.",
     "level": "verified",
     "sources": [
      "s16",
      "s7",
      "s9",
      "s10"
     ],
     "status": "open",
     "note": "The 89.37% figure is a user report (reported level), not confirmed by X-Humanoid.",
     "mitigation": {
      "text": "RoboMIND-dataset-utils (2026-04-14) provides check scripts, per-file CSV reports and a joint-to-pose script.",
      "sources": [
       "s16"
      ]
     },
     "short": "X-Humanoid's own checks found that end-effector values (the position and angle of the robot's gripper or hand) stay almost constant in three subsets. Users report further problems."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The digital twin check is small and cannot be repeated",
     "text": "Table VII compares 2 policies on 5 Franka tasks with 10 trials per task; each Pearson r rests on 5 points and has no interval. The twin scored policies far lower than the real robot (ACT 1.4/10 vs 5.4/10 on average). In Figure 17 a policy trained only on twin data scored 0.9 in the twin and 0.1 on the real robot, which the authors attribute to physics gaps in contact-rich tasks. The version 1 twin is not released, so others cannot rerun the check.",
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "status": "open",
     "short": "The digital twin was checked with 2 policies on 5 tasks. The twin is not released, so others cannot repeat the check."
    },
    {
     "id": "i5",
     "type": "other",
     "title": "The backgrounds are simple and fixed",
     "text": "The authors list relatively simple background environments as a limitation. In their generalisation test, three unseen tablecloths cut FR-PlaceBreadPlate success to 0-2 of 10 for OpenVLA, RDT-1B and CrossFormer (from 4, 9 and 10 of 10).",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "In the authors' test, three new tablecloths cut success on one task to 0 to 2 of 10 trials."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "The name is shared with RoboMIND 2.0",
     "text": "The version 1 card and site announce 'RoboMIND V2.0', a different and larger dataset (over 310K trajectories, six robot types). Chinese news coverage uses 'RoboMIND' for 2.0 figures (300,000+ trajectories, 700+ tasks, over 20 million downloads). Sizes and download counts are easy to mix up.",
     "level": "verified",
     "sources": [
      "s7",
      "s22",
      "s26"
     ],
     "status": "open",
     "short": "The name 'RoboMIND' is often used for RoboMIND 2.0, a separate and larger dataset. Their sizes and download counts are easy to mix up."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "RoboMIND is mainly training data. Its paper's success rates are one-off tests by the builders on their own robots and task setups, with 10 trials each and no error bars. They are not comparable with results in other papers.",
     "basis": [
      "facts.kind",
      "facts.evaluator",
      "facts.leaderboard",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "RoboMIND is mainly training data. Its test results are one-off tests by its builders."
    },
    {
     "id": "r2",
     "text": "The digital-twin check is small. Two policies on five tasks, with ten trials each, cannot show that the twin ranks many policies correctly. The twin also scored policies far lower than the real robot. A policy trained only in the twin scored well there and failed on the real robot.",
     "basis": [
      "facts.sim_to_real",
      "issues.i4"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "The check of the digital twin is too small to show that the twin can be used as a test."
    },
    {
     "id": "r3",
     "text": "When quoting its size, name the version and say whether simulation is included. Version 1.0 has 55k trajectories, and versions 1.1 and 1.2 have 107k, of which about 30k are simulated. RoboMIND 2.0, a different dataset, has over 310K.",
     "basis": [
      "facts.demonstrations",
      "facts.version",
      "issues.i1",
      "issues.i2",
      "issues.i6"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "When quoting its size, name the version and say whether simulated data is counted."
    }
   ],
   "searched": [
    {
     "for": "validity",
     "where": "arXiv v1, v2 and v3 full text (the correlation table first appears in v3; the co-training figure in v2), RSS 2025 PDF pages 1-2 and 11-15, project site, dataset card, ModelScope card and file tree, X-Humanoid GitHub organisation (RoboMIND-Sim and x-humanoid-vla-simulation-benchmark belong to RoboMIND 2.0 or other projects). No independent sim-versus-real study of RoboMIND was found; papers that use the data train on it and do not report RoboMIND scores.",
     "date": "2026-10-10"
    },
    {
     "for": "count conflict (per-robot trajectories)",
     "where": "All three arXiv versions, RSS PDF Figure 1, Hugging Face and ModelScope cards, ModelScope folder listing for benchmark1_0, 1_1 and 1_2, all_robot_h5_info_v1.2.md, RoboMIND_v1_2_instr.csv. Archives are split tar.gz parts, so per-folder episode counts were not verified.",
     "date": "2026-10-10"
    },
    {
     "for": "license_code and license_assets",
     "where": "Paper link to x-humanoid-training-toolchain (redirects to Open-X-Humanoid), its LICENSE file, RoboMIND-dataset-utils LICENSE, website repo LICENSE, dataset card, ModelScope metadata.",
     "date": "2026-10-10"
    },
    {
     "for": "official Chinese announcement",
     "where": "X-Humanoid official site home and news pages (the open-source portal is script-rendered); Beijing News report of 2026-09-02 (secondary). The shared web-search budget ran out before an official RoboMIND v1 launch article was found.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RoboMIND (arXiv abstract page, submission history v1 to v3)",
     "url": "https://arxiv.org/abs/2412.13877",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RoboMIND full text v3 (RSS 2025 version)",
     "url": "https://arxiv.org/html/2412.13877v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "RoboMIND full text v1",
     "url": "https://arxiv.org/html/2412.13877v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "RoboMIND full text v2",
     "url": "https://arxiv.org/html/2412.13877v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RoboMIND, Robotics: Science and Systems XXI proceedings page (paper 152)",
     "url": "https://www.roboticsproceedings.org/rss21/p152.html",
     "type": "paper",
     "publisher": "RSS 2025",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "RoboMIND, RSS 2025 PDF (Figure 1, Table VII, Figure 17)",
     "url": "https://www.roboticsproceedings.org/rss21/p152.pdf",
     "type": "paper",
     "publisher": "RSS 2025",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "x-humanoid-robomind/RoboMIND dataset card (composition, version notes, data notes)",
     "url": "https://huggingface.co/datasets/x-humanoid-robomind/RoboMIND",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2026-04-14",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Hugging Face Hub API: x-humanoid-robomind/RoboMIND (licence, gated, downloads, lastModified)",
     "url": "https://huggingface.co/api/datasets/x-humanoid-robomind/RoboMIND?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated&expand[]=lastModified&expand[]=createdAt&expand[]=cardData",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "RoboMIND discussion #6: UR5e data with almost no movement (user reports, open)",
     "url": "https://huggingface.co/datasets/x-humanoid-robomind/RoboMIND/discussions/6",
     "type": "repo",
     "publisher": "Hugging Face (user reports)",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "RoboMIND discussion #8: invalid end-effector poses in h5_ur_1rgb (user report, open)",
     "url": "https://huggingface.co/datasets/x-humanoid-robomind/RoboMIND/discussions/8",
     "type": "repo",
     "publisher": "Hugging Face (user report)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "ModelScope API: X-Humanoid/RoboMIND (licence, downloads, dates)",
     "url": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND",
     "type": "index",
     "publisher": "ModelScope",
     "date": "2026-07-20",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "ModelScope file tree: X-Humanoid/RoboMIND (benchmark1_0, 1_1, 1_2 folders)",
     "url": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND/repo/tree?Revision=master&Root=benchmark1_1_compressed&Recursive=false",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "RoboMIND all_robot_h5_info_v1.2.md (file formats incl. Simulation Franka and Simulation Tien Kung)",
     "url": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND/repo?Revision=master&FilePath=static/all_robot_h5_info_v1.2.md",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "RoboMIND_v1_2_instr.csv (task instruction list)",
     "url": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND/repo?Revision=master&FilePath=static/RoboMIND_v1_2_instr.csv",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Open-X-Humanoid/RoboMIND-dataset-utils README and LICENSE",
     "url": "https://github.com/Open-X-Humanoid/RoboMIND-dataset-utils",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2026-04-14",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Open-X-Humanoid/x-humanoid-training-toolchain (README, LICENSE, GitHub API record)",
     "url": "https://github.com/Open-X-Humanoid/x-humanoid-training-toolchain",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2026-09-16",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "RoboMIND project site",
     "url": "https://x-humanoid-robomind.github.io/",
     "type": "site",
     "publisher": "RoboMIND team",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "GitHub API: x-humanoid-robomind/x-humanoid-robomind.github.io (stars, licence)",
     "url": "https://api.github.com/repos/x-humanoid-robomind/x-humanoid-robomind.github.io",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "北京人形机器人创新中心有限公司 official site (home page)",
     "url": "https://x-humanoid.com",
     "type": "site",
     "publisher": "Beijing Innovation Center of Humanoid Robotics",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Semantic Scholar API record for arXiv:2412.13877",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2412.13877?fields=title,citationCount,influentialCitationCount,venue,publicationDate,externalIds",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "RoboMIND 2.0 full text v3 (introduction and Table 1)",
     "url": "https://arxiv.org/html/2512.24653v3",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Open-X-Humanoid/RoboMIND-Sim README (RoboMIND 2.0 Isaac Sim benchmark)",
     "url": "https://github.com/Open-X-Humanoid/RoboMIND-Sim",
     "type": "repo",
     "publisher": "X-Humanoid",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "EO-1: An Open Unified Embodied Foundation Model for General Robot Control (training data table)",
     "url": "https://arxiv.org/html/2508.21112",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "XR-1: Towards Versatile Vision-Language-Action Models via Learning Unified Vision-Motion Representations (pretraining data)",
     "url": "https://arxiv.org/html/2511.02776",
     "type": "paper",
     "publisher": "arXiv (X-Humanoid)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "加速打造全球具身智能数据平台 北京人形开源数据集下载量破2000万",
     "url": "https://www.bjnews.com.cn/detail/1788347509129990.html",
     "type": "secondary",
     "publisher": "新京报 (Beijing News)",
     "date": "2026-09-02",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Expanded to a full entry. Resolved the per-robot count conflict at the source as far as possible (paper text vs Figure 1 and card; twin data folded into robot totals; one AgileX figure unresolved). Added the code licence (Apache-2.0 toolchain), builder legal form, data-quality findings from X-Humanoid's own check scripts, two author-run sim-versus-real comparisons, and the relation to RoboMIND 2.0."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "Whether a policy would get the same score in another lab.",
     "sub": "All tests in the paper ran on X-Humanoid's own robot setups.",
     "basis": [
      "facts.real_reproducibility",
      "facts.evaluator"
     ]
    },
    {
     "id": "l2",
     "text": "How well a policy does in new surroundings.",
     "sub": "In the one task tested this way, new tablecloths cut success to 0 to 2 of 10 trials.",
     "basis": [
      "facts.generalisation",
      "issues.i5"
     ]
    },
    {
     "id": "l3",
     "text": "Whether the digital twin (the simulated copy) predicts real-robot results.",
     "sub": "The authors checked it with only 2 policies on 5 tasks.",
     "basis": [
      "facts.sim_to_real",
      "issues.i4"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "RoboMIND digital-twin check (Table VII)",
     "date": "2025-05",
     "by": "authors",
     "method": "ACT and Diffusion Policy, trained on real data, were each run 10 times on the same 5 Franka tasks in the Isaac Sim twin and on the real robot.",
     "result": "Pearson correlation r = 0.83 (ACT) and 0.91 (Diffusion Policy), each across 5 tasks",
     "authors_view": "positive correlations",
     "n_policies": 2,
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "note": "First appears in arXiv v3 (2025-05-27) and the RSS 2025 version. Average successes: ACT 1.4/10 twin vs 5.4/10 real; Diffusion Policy 3.2/10 vs 8.0/10."
    },
    {
     "id": "v2",
     "name": "RoboMIND co-training study (Figure 17)",
     "date": "2025-02",
     "by": "authors",
     "method": "7 ACT policies trained on different mixes of real and twin data were each scored on one task (FR-UprightBlueCup) in the twin and on the real robot.",
     "result": "No statistic was reported. The policy trained only on twin data scored 0.9 in the twin and 0.1 on the real robot. The policy trained only on real data scored 0 in the twin and 0.6 on the real robot.",
     "authors_view": "simulation data alone is insufficient",
     "n_policies": 7,
     "level": "verified",
     "sources": [
      "s4",
      "s6"
     ],
     "note": "The study tests data mixes, not the twin as a test, but it scores the same policies in both places. Bar values read from Figure 17 of the RSS PDF: real 0.6, 0.7, 0.9, 0.9, 1.0, 1.0, 0.1 and twin 0, 0.2, 0.4, 0.4, 0.7, 0.9, 0.9 for real:twin ratios 100:0, 100:100, 100:200, 100:300, 100:400, 100:500, 0:500."
    }
   ]
  },
  {
   "id": "robomind-2-0",
   "name": "RoboMIND 2.0",
   "aliases": [
    "RoboMIND V2.0",
    "RoboMIND2.0",
    "RoboMIND-Sim (simulation benchmark companion)",
    "MIND-2 (model)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Real-robot dataset that ships a re-runnable simulation benchmark (RoboMIND-Sim: Isaac Sim tasks with standardized evaluation scripts). The paper also reports real-robot evaluations.",
   "summary": {
    "text": "Dual-arm and mobile manipulation demonstrations on six robot types, with tactile data and a small Isaac Sim benchmark.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Beijing Innovation Center of Humanoid Robotics; State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Corresponding authors Shanghang Zhang (PKU) and Jian Tang (X-Humanoid)."
    },
    "builder_type": {
     "value": "robot-company",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Same lead organisation as RoboMIND; legal form not checked."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-12 (arXiv v1, 31 Dec 2025); ModelScope dataset created 2026-01-04",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "ModelScope dates from https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND2.0"
    },
    "latest_update": {
     "value": "ModelScope LastUpdatedTime 2026-07-15; arXiv v3 2026-02-27; RoboMIND-Sim created 2026-03-04, last push 2026-03-26",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "RoboMIND-Sim dates from https://api.github.com/repos/Open-X-Humanoid/RoboMIND-Sim"
    },
    "version": {
     "value": "arXiv v3 (2026-02-27); ModelScope X-Humanoid/RoboMIND2.0",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "arXiv preprint only (no journal_ref or venue comment as of 2026-10-10)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "bimanual",
      "mobile-manipulation",
      "manipulation",
      "dexterous",
      "long-horizon",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "bimanual-arm",
      "mobile-manipulator",
      "humanoid",
      "dexterous-hand"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "home",
      "kitchen",
      "retail-logistics",
      "industrial",
      "office-lab"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Over 310K trajectories; over 1,000 h; 129 skills; 1,139 objects; 12K tactile; 20K mobile; 20K simulated trajectories (Franka dual-arm and Tien Kung)",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "ModelScope README repeats 310K, 1,000 h, 12K tactile, 20K mobile, 6 embodiments."
      },
      {
       "value": "Task-count conflict inside the paper: 739 tasks (abstract) vs 759 tasks (introduction and contributions)",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "RoboMIND-Sim details from https://github.com/Open-X-Humanoid/RoboMIND-Sim Possibly different Isaac Sim versions (README mentions data collected in Isaac Sim 4.5 and code for 5.1). Not reconciled by the authors."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: no leaderboard; the RoboMIND-Sim README shows only the authors' ACT numbers)"
    },
    "license_code": {
     "value": "unknown: RoboMIND-Sim repo has no LICENSE (GitHub license API returns null)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Apache-2.0 (ModelScope dataset metadata: 'Apache License 2.0')",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "access": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "unclear: ModelScope metadata shows ApprovalMode 1 and ProtectedMode 2; we could not confirm what these mean",
     "note": "We did not try to download."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Tried, not measured (: real-robot gains from adding sim training data (co-training ratios 1:0, 0:1, 1:1, 1:5); no correlation statistic between sim and real evaluation)",
     "note": "RoboMIND-Sim (4 Tien Kung tasks) has no paired sim/real evaluation correlation in the paper. The paper evaluates policies trained only in sim inside sim (Table 8). Separately, it runs policies co-trained with real and sim data on the real robots. Adding more sim data (1:1 to 1:5) raised real success on some tasks (e.g. Diffusion Policy Task 3 0.6 to 0.8). That shows sim training data transfers. It does not show that sim scores predict real scores."
    },
    "kind": {
     "value": "dataset",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboMIND 2.0: A Multimodal, Bimanual Mobile Manipulation Dataset for Generalizable Embodied Intelligence",
     "url": "https://arxiv.org/abs/2512.24653",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s2": {
     "title": "RoboMIND 2.0: A Multimodal, Bimanual Mobile Manipulation Dataset for Generalizable Embodied Intelligence (full text)",
     "url": "https://arxiv.org/html/2512.24653v3",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s3": {
     "title": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND2.0",
     "url": "https://modelscope.cn/api/v1/datasets/X-Humanoid/RoboMIND2.0",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "Open-X-Humanoid/RoboMIND-Sim on GitHub (repository)",
     "url": "https://github.com/Open-X-Humanoid/RoboMIND-Sim",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "Open-X-Humanoid/RoboMIND-Sim on GitHub (repository)",
     "url": "https://api.github.com/repos/Open-X-Humanoid/RoboMIND-Sim",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2512.24653",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "robothor-objectnav",
   "name": "RoboTHOR ObjectNav",
   "aliases": [
    "RoboTHOR",
    "RoboTHOR Challenge",
    "RoboTHOR Object Navigation Challenge"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores navigation agents in simulated apartments that have physical counterparts; ran as a public challenge.",
   "summary": {
    "text": "Object-goal navigation benchmark in 89 simulated apartments, 14 of them rebuilt physically for robot testing.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Allen Institute for AI (13 authors; repo allenai/robothor-challenge)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "first_release": {
     "value": "2020-04 (arXiv v1 2020-04-14)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "CVPR 2020 (arXiv comment)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "2021 RoboTHOR ObjectNav Challenge: announced 2021-02-17, submissions closed 2021-05-31, winners 2021-06-19; repo last pushed 2021-05-17",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The 2021 challenge page states it ran only in simulation due to COVID-19. Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Visual semantic navigation to an object category from egocentric RGB-D. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "mobile-base"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "robots": {
     "value": "LoCoBot with Intel RealSense RGB-D camera",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "home"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Apartments. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "path-efficiency",
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Headline SPL. Success: target within 1 m, agent issues STOP, object visible in final frame. Checked 2026-10-10."
    },
    "top_score": {
     "value": "Challenge page: best pretrained AllenAct baseline about 26% test success. Final challenge results not found (leaderboard host unreachable).",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "official",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Official (historical): leaderboard.allenai.org/robothor_objectnav; host did not resolve (DNS failure) on 2026-10-10)",
     "note": "Tried WebFetch and curl. Checked 2026-10-10."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "GitHub API licence detection for allenai/robothor-challenge. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "Tried, not measured (: 1 learned model on LoCoBot; Image model Easy 55.30 success / 38.12 SPL in sim vs 33.33 / 3.53 real; Medium 28.79/19.12 vs 16.66/3.70; Hard 1.47/0.97 vs 0.00/0.00. Paper reports a 'significant gap'; no correlation statistic.)",
     "note": "Paper ran one learned model (Image) on a LoCoBot in the physical apartments: Easy success 55.30 sim vs 33.33 real, SPL 38.12 vs 3.53. No correlation statistic; 2021 challenge ran in simulation only (COVID-19)."
    },
    "citations": {
     "value": 340,
     "display": "340",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No challenge after 2021 found; repo last pushed 2021-05. Checked 2026-10-10."
    },
    "version": {
     "value": "2021 challenge configuration; README pins ai2thor==2.7.2 for AllenAct baselines",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "license_data": {
     "value": "Episodes ship in the Apache-2.0 challenge repo; no separate dataset licence found",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboTHOR: An Open Simulation-to-Real Embodied AI Platform",
     "url": "https://arxiv.org/abs/2004.06799",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2020-04"
    },
    "s2": {
     "title": "allenai/robothor-challenge on GitHub (repository)",
     "url": "https://github.com/allenai/robothor-challenge",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s3": {
     "title": "https://ai2thor.allenai.org/robothor/cvpr-2021-challenge/",
     "url": "https://ai2thor.allenai.org/robothor/cvpr-2021-challenge/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "RoboTHOR: An Open Simulation-to-Real Embodied AI Platform (full text)",
     "url": "https://arxiv.org/html/2004.06799",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2020-04"
    },
    "s5": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2004.06799",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "robotwin-2-0",
   "name": "RoboTwin 2.0",
   "full_name": "RoboTwin 2.0: A Scalable Data Generator and Benchmark with Strong Domain Randomization for Robust Bimanual Robotic Manipulation",
   "aliases": [
    "RoboTwin",
    "RoboTwin2.0",
    "RoboTwin 2.0 Leaderboard",
    "RoboTwin-OD"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated two-arm tasks with a fixed evaluation protocol (100 trials per task in a clean and a randomized setting) and an official leaderboard.",
   "summary": {
    "text": "RoboTwin 2.0 is a simulated benchmark and data generator for two-arm robots from Shanghai Jiao Tong University, HKU MMLab and partners: 50 tabletop tasks scored by success rate in clean scenes (Easy) and randomized scenes (Hard), with an official leaderboard. The official protocol trains on 50 clean demonstrations per task, but many papers use a second protocol that adds 500 randomized demonstrations per task, and the two give very different Hard scores.",
    "sources": [
     "s2",
     "s11",
     "s34"
    ],
    "short": "RoboTwin 2.0 is a simulated benchmark and data generator for two-armed robots. It tests robot policies (the models that control a robot) on 50 tabletop tasks in clean and randomized scenes, and it has an official leaderboard."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents a benchmark with a fixed protocol; the project runs an official leaderboard."
    },
    "kind_secondary": {
     "value": [
      "dataset",
      "platform"
     ],
     "display": "Also a data generator, a 100,000-trajectory dataset and a simulation platform built on SAPIEN",
     "level": "verified",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "2.0",
     "display": "No release tags. Paper arXiv v1 2025-06-22, v2 2025-08-27; ICML 2026. Evaluation moved to the XPolicyLab interface on 2026-08-03.",
     "level": "verified",
     "sources": [
      "s1",
      "s5",
      "s4"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RoboTwin 1.0 and early version",
       "display": "Earlier benchmarks by the same group: 'RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins', CVPR 2025 (arXiv 2504.13059), and an early version at an ECCV 2024 workshop (arXiv 2409.02920). Kept on separate branches.",
       "level": "verified",
       "sources": [
        "s5",
        "s50"
       ]
      },
      {
       "value": "Leaderboard changes",
       "display": "Launched 2025-08-06 with single-task baselines; co-train and single-task results merged into one board 2026-07-22; default ranking set to the mean of Easy and Hard 2026-08-27.",
       "level": "verified",
       "sources": [
        "s5",
        "s11"
       ]
      },
      {
       "value": "Data and reproducibility updates",
       "display": "Official demo_clean training data refreshed 2026-09-14 (standard RGB images, 'legal-joint replacements'); seed reproducibility fixes merged 2026-09-24.",
       "level": "verified",
       "sources": [
        "s15",
        "s17"
       ]
      }
     ],
     "short": "2.0. There are no release tags."
    },
    "publishers": {
     "value": [
      "Shanghai Jiao Tong University",
      "HKU MMLab"
     ],
     "display": "26 authors. Equally leading organisations: SJTU (MoE Key Lab of AI, AI Institute) and HKU MMLab. Other affiliations include Lumina EAI, Shanghai AI Lab, Shenzhen University, NJU, TeleAI, Fudan, SYSU, USTC, CSU, SUSTech, Tsinghua, D-Robotics, NEU and HKU-SH ICRC.",
     "level": "verified",
     "sources": [
      "s1",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Last authors Ping Luo and Yao Mu. The licence and leaderboard contact is Tianxing Chen.",
     "short": "Shanghai Jiao Tong University and HKU MMLab, with 26 authors"
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Mostly universities and public labs; D-Robotics and Lumina EAI appear as co-affiliations."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2025-06",
     "display": "Released 2025-06-21 (README); arXiv v1 2025-06-22",
     "level": "verified",
     "sources": [
      "s1",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "June 2025"
    },
    "latest_update": {
     "value": "2026-10",
     "display": "Leaderboard updated 2026-10-10; last code push 2026-09-24; Hugging Face data changed 2026-09-22",
     "level": "verified",
     "sources": [
      "s11",
      "s6",
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "October 2026"
    },
    "published_at": {
     "value": "ICML 2026",
     "display": "Proceedings of the 43rd International Conference on Machine Learning, PMLR 306:13673-13699",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "bimanual",
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Language instructions are one of the five randomized factors; the Hard setting uses unseen instructions."
    },
    "generalisation": {
     "value": [
      "visual",
      "scene-layout",
      "language",
      "object-pose"
     ],
     "display": "Hard setting: random backgrounds, clutter, lighting, table height (up to 3 cm) and unseen instructions. Objects start in new places in every episode.",
     "level": "verified",
     "sources": [
      "s10",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Official configs: demo_clean (Easy) evaluates with seen instructions and no randomization; demo_randomized (Hard) uses unseen instructions, random backgrounds (2% clean), cluttered table, table height up to 0.03 m and random light. Under the official protocol policies train on clean data only, so everything randomized is unseen in training.",
     "short": "Random clutter, lighting and backgrounds, and unseen instructions, in the Hard setting"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Real-robot runs in the paper test whether its synthetic data helps real policies; they are not a benchmark track."
    },
    "simulator": {
     "value": "SAPIEN 3.0.0b1",
     "display": "SAPIEN 3.0.0b1 physics and rendering; mplib 0.2.1 and cuRobo motion planning",
     "level": "verified",
     "sources": [
      "s8",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "SAPIEN 3.0"
    },
    "embodiment": {
     "value": [
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Data and tasks support five dual-arm platforms; the benchmark and leaderboard score only Aloha-AgileX."
    },
    "robots": {
     "value": "Aloha-AgileX (scored); data also for ARX-X5, Franka, UR5, Piper",
     "display": "Benchmark robot: Aloha-AgileX dual arm (simulated). The generator supports Aloha-AgileX, Piper, Franka, UR5 and ARX-X5 pairs.",
     "level": "verified",
     "sources": [
      "s2",
      "s11",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Per-task data archives cover 3 to 5 robots (for example, handover_block has no Franka or UR5 data; adjust_bottle has no Piper data), matching the paper's lower expert success on some robots.",
     "short": "Scores use only the Aloha-AgileX robot"
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Read from the task list and randomization settings."
    },
    "tasks": {
     "value": 50,
     "level": "verified",
     "sources": [
      "s1",
      "s11"
     ],
     "checked": "2026-10-10",
     "short": "50 tasks"
    },
    "objects": {
     "value": 731,
     "display": "731 object instances in 147 categories (RoboTwin-OD): 534 made in-house with the Rodin 3D generator, 153 from Objaverse, 44 articulated objects from PartNet-Mobility",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "731 objects"
    },
    "demonstrations": {
     "value": "over 100,000",
     "display": "Over 100,000 scripted expert trajectories over 50 tasks. Per task and robot: 50 clean and 500 randomized.",
     "level": "verified",
     "sources": [
      "s2",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Demonstrations come from task programs written with an LLM and checked in simulation, not from human teleoperation. 11,000 background textures generated with Stable Diffusion v2 and filtered by people.",
     "short": "Over 100,000 scripted demonstrations"
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Success is a task-specific predicate checked by the simulator; an episode counts as a success the first time the predicate holds."
    },
    "metric_detail": {
     "value": "success rate, Easy and Hard",
     "display": "Success rate per task over 100 trials, in Easy (clean) and Hard (randomized). The leaderboard ranks by the mean of the Easy and Hard averages.",
     "level": "verified",
     "sources": [
      "s11",
      "s12",
      "s9"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Official protocol",
       "display": "Train on 50 clean demonstrations per task (2,500 in total) on Aloha-AgileX. Co-train track: one policy for all 50 tasks; Single track: one checkpoint per task.",
       "level": "verified",
       "sources": [
        "s11",
        "s12"
       ]
      },
      {
       "value": "Second protocol (50 clean + 500 randomized)",
       "display": "Introduced by Motus (2025-12): one policy trained on 2,500 clean plus 25,000 randomized demonstrations. Used by Fast-WAM, MotuBrain and others and by the 2026 audit. Not on the official board.",
       "level": "verified",
       "sources": [
        "s34",
        "s35",
        "s36",
        "s30"
       ]
      },
      {
       "value": "Test episodes the expert can solve",
       "display": "For each candidate seed (starting at 100000), the scripted expert runs first; seeds where it fails or the scene is unstable are skipped. Only expert-solvable scenes are scored.",
       "level": "verified",
       "sources": [
        "s9"
       ]
      }
     ],
     "short": "Success rate in the Easy and Hard settings"
    },
    "trials": {
     "value": 100,
     "display": "100 per task in each setting",
     "level": "verified",
     "sources": [
      "s2",
      "s11"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Two board entries look like 50-trial runs",
       "display": "All 100 per-task values of GigaBrain-0.7 and of Discrete Forcing are even numbers; every other entry has odd values. This fits 50 trials per task, not 100.",
       "level": "inferred",
       "sources": [
        "s11",
        "s37"
       ],
       "note": "Counted by us from the leaderboard data file. GigaBrain-0.7's report does not state its trial count."
      }
     ],
     "short": "100 per task in each setting"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s11",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper and the leaderboard give single success rates without error bars or seed spread."
    },
    "evaluator": {
     "value": "both",
     "level": "verified",
     "sources": [
      "s11",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "Of 24 board entries, 16 list the RoboTwin Team as the team that produced the numbers and 8 list outside teams. The page says results are reproduced via the XPolicyLab interface and requires public code, weights and a technical report. Numbers in papers are self-reported."
    },
    "leaderboard": {
     "value": "official",
     "display": "Official: 24 entries (19 co-train, 5 single-task), updated 2026-10-10",
     "level": "verified",
     "sources": [
      "s11",
      "s12"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": 79.14,
     "display": "79.14 mean of Easy and Hard (PatchWAM, 2026-10): Easy 91.56, Hard 66.72, official 50-clean-demo protocol. Best Hard: ME-Dex-1.0, 68.12. Papers using the 50+500 protocol report up to 95.8 / 96.1; those are not comparable.",
     "level": "verified",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "Easy and Hard values are read from the leaderboard data file; the mean follows the board's own ranking rule ((Easy + Hard) / 2), computed by us. Dates are board listing dates.",
     "items": [
      {
       "value": 24.11,
       "display": "RDT (single-task), 2025-08: Easy 34.5, Hard 13.72",
       "level": "verified",
       "sources": [
        "s11",
        "s2"
       ],
       "data": {
        "model": "RDT",
        "date": "2025-08",
        "avg": 24.11,
        "easy": 34.5,
        "hard": 13.72,
        "rl": false
       }
      },
      {
       "value": 30.1,
       "display": "DP3 (single-task), 2025-08: Easy 55.24, Hard 4.96",
       "level": "verified",
       "sources": [
        "s11",
        "s2"
       ],
       "note": "The paper says DP3's results 'partly stem from perfect point clouds and clean background segmentation in simulation' (issues.i8).",
       "data": {
        "model": "DP3",
        "date": "2025-08",
        "avg": 30.1,
        "easy": 55.24,
        "hard": 4.96,
        "rl": false
       }
      },
      {
       "value": 31.38,
       "display": "π0 (single-task), 2025-08: Easy 46.42, Hard 16.34",
       "level": "verified",
       "sources": [
        "s11",
        "s2"
       ],
       "data": {
        "model": "π0",
        "date": "2025-08",
        "avg": 31.38,
        "easy": 46.42,
        "hard": 16.34,
        "rl": false
       }
      },
      {
       "value": 44.45,
       "display": "X-VLA (co-train, run by RoboTwin Team), 2026-08: Easy 68.0, Hard 20.9",
       "level": "verified",
       "sources": [
        "s11"
       ],
       "note": "X-VLA's own paper reports 70.0 / 39.0 (issues.i2).",
       "data": {
        "model": "X-VLA",
        "date": "2026-08",
        "avg": 44.45,
        "easy": 68,
        "hard": 20.9,
        "rl": false
       }
      },
      {
       "value": 58.35,
       "display": "π0.5 (co-train, run by RoboTwin Team), 2026-08: Easy 70.7, Hard 46.0",
       "level": "verified",
       "sources": [
        "s11"
       ],
       "data": {
        "model": "π0.5",
        "date": "2026-08",
        "avg": 58.35,
        "easy": 70.7,
        "hard": 46,
        "rl": false
       }
      },
      {
       "value": 67.35,
       "display": "GigaBrain-0.7 (GigaAI), 2026-08: Easy 66.8, Hard 67.9",
       "level": "verified",
       "sources": [
        "s11",
        "s37"
       ],
       "note": "Probably 50 trials per task (facts.trials).",
       "data": {
        "model": "GigaBrain-0.7",
        "date": "2026-08",
        "avg": 67.35,
        "easy": 66.8,
        "hard": 67.9,
        "rl": false
       }
      },
      {
       "value": 71.34,
       "display": "UniWAM (OLA-HKUSTGZ), 2026-09: Easy 75.12, Hard 67.56",
       "level": "verified",
       "sources": [
        "s11"
       ],
       "data": {
        "model": "UniWAM",
        "date": "2026-09",
        "avg": 71.34,
        "easy": 75.12,
        "hard": 67.56,
        "rl": false
       }
      },
      {
       "value": 78.85,
       "display": "ME-Dex-1.0 (Li Auto), 2026-09: Easy 89.58, Hard 68.12",
       "level": "verified",
       "sources": [
        "s11"
       ],
       "data": {
        "model": "ME-Dex-1.0",
        "date": "2026-09",
        "avg": 78.85,
        "easy": 89.58,
        "hard": 68.12,
        "rl": false
       }
      },
      {
       "value": 79.14,
       "display": "PatchWAM (PatchWAM Team), 2026-10: Easy 91.56, Hard 66.72",
       "level": "verified",
       "sources": [
        "s11"
       ],
       "data": {
        "model": "PatchWAM",
        "date": "2026-10",
        "avg": 79.14,
        "easy": 91.56,
        "hard": 66.72,
        "rl": false
       }
      },
      {
       "value": "95.8 / 96.1 (other protocol)",
       "display": "MotuBrain, 2026-04: clean 95.8, randomized 96.1, trained on 50 clean + 500 randomized demonstrations per task. Not comparable with the rows above.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "91.88 / 91.78 (other protocol)",
       "display": "Fast-WAM, 2026-03: clean 91.88, randomized 91.78 under the 50+500 protocol. The same model scored 77.8 / 1.9 on the official board (issues.i1).",
       "level": "verified",
       "sources": [
        "s35",
        "s11"
       ]
      }
     ],
     "short": "79.14 average of Easy and Hard (PatchWAM, October 2026)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE: MIT, 'Copyright (c) 2025 Tianxing Chen'."
    },
    "license_data": {
     "value": "MIT",
     "display": "MIT on the Hugging Face dataset card (metadata only; the card has no other text)",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Apache-2.0 (third-party copy)",
       "display": "LeRobot's re-hosted copy lerobot/robotwin_unified is described as Apache 2.0 in LeRobot's docs.",
       "level": "verified",
       "sources": [
        "s24"
       ],
       "note": "A third party's label; it does not change RoboTwin's own licence."
      }
     ],
     "short": "MIT, as labelled on the dataset card"
    },
    "license_assets": {
     "value": "unclear",
     "display": "The MIT-labelled data repo bundles 44 PartNet-Mobility objects, whose source terms allow non-commercial research and education only, and 153 Objaverse objects, which carry per-object Creative Commons licences, some non-commercial. No licence is stated for the 534 in-house Rodin models.",
     "level": "inferred",
     "sources": [
      "s2",
      "s14",
      "s45",
      "s46"
     ],
     "checked": "2026-10-10",
     "note": "Objaverse card: ODC-By 1.0 for the collection; objects under CC-BY (721K), CC-BY-NC (25K), CC-BY-NC-SA (52K), CC-BY-SA (16K) or CC0 (3.5K). Which licences RoboTwin's 153 objects carry was not checked. Not legal advice."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s13",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Code on GitHub; data, assets and co-train checkpoints in an ungated Hugging Face repo.",
     "short": "Open to download"
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s7",
      "s13",
      "s45",
      "s46"
     ],
     "checked": "2026-10-10",
     "note": "Code and data labels (MIT) allow commercial use, but bundled third-party 3D assets come with non-commercial source terms (facts.license_assets). Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Tried, not measured: policies trained with RoboTwin data ran on a real robot, but no study compares the same policies' RoboTwin 2.0 scores with real scores.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s38",
      "s48"
     ],
     "checked": "2026-10-10",
     "note": "Data-transfer evidence only. The paper trained RDT on a real COBOT-Magic robot for 4 tasks: with 10 real demonstrations plus 1,000 RoboTwin trajectories, average success on unseen cluttered backgrounds rose from 9.0% to 42.0%; with synthetic data only it reached 29.5% (arXiv: 367% and 228% relative gains; ICML abstract: '3.6x' and '2.2x'). Trial counts for these real runs are not stated. Independent check on the earlier version only: WorldEval (2025-05) ran real-trained policies in RoboTwin 1.0 on 3 custom tasks, with images translated by MidJourney, and found average Pearson r 0.411 and MMRV 0.261 against real results. That study predates 2.0 and uses other tasks, so it is not counted as validity for 2.0. A May 2026 paper by other authors also marks RoboTwin as having no validated sim-to-real correlation in its comparison table.",
     "short": "Policies trained with its data have run on a real robot. No study checks its scores against real results."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "independent_audit": {
     "value": "fewer failures than LIBERO, CALVIN and SimplerEnv",
     "display": "The 2026 benchmark audit (TTIC, UChicago, Argonne) found no shortcut on RoboTwin 2.0, and 73.7% of 19 claimed new bests were provably significant.",
     "level": "verified",
     "sources": [
      "s30",
      "s31",
      "s32"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Shortcut probe",
       "display": "A 0.09B DINOv2 + MLP probe with a task embedding instead of language scored 60.4% clean and 59.4% randomized under the 50+500 protocol, against 95.8% and 96.1% for the best reported model.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "Significance",
       "display": "Of 19 previous-best-to-new comparisons on the 50+500 protocol (Hard), 73.7% are provably significant, 15.8% inconclusive and 10.5% show no improvement. The audit notes fewer reported results make larger gaps easier to prove.",
       "level": "verified",
       "sources": [
        "s30",
        "s32",
        "s31"
       ]
      }
     ],
     "short": "A 2026 audit found no shortcut to a high score. Most claimed gains it checked could be shown to be statistically significant."
    },
    "challenges": {
     "value": [
      "RoboTwin Dual-Arm Collaboration Challenge (CVPR 2025)",
      "Challenge Cup 2025 dual-arm topic",
      "GigaBrain Challenge 2026 RoboTwin track"
     ],
     "level": "verified",
     "sources": [
      "s39",
      "s40",
      "s41"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RoboTwin Dual-Arm Collaboration Challenge (CVPR 2025 MEIS Workshop)",
       "display": "Simulation round 1 on RoboTwin 1.0, round 2 on RoboTwin 2.0 (6 tasks, randomized, unseen instructions, 100 trials per task), then a real-robot round on 5 different tasks. 64 teams, over 400 participants. Best real-world score 26.4 of 100 points.",
       "level": "verified",
       "sources": [
        "s39"
       ]
      },
      {
       "value": "Challenge Cup 2025 (挑战杯 '人工智能+' topic 4)",
       "display": "Dual-arm algorithm deployable on edge devices: 5 RoboTwin tasks scored online in simulation on D-Robotics' cloud, models up to 1B parameters, final round on real robots and the RDK S100 board. Submission deadline 2025-08-17.",
       "level": "verified",
       "sources": [
        "s40"
       ],
       "note": "The problem statement (2025-05-14) predates the 2.0 release and links the RoboTwin repository."
      },
      {
       "value": "GigaBrain Challenge 2026 (CVPR 2026)",
       "display": "Track 1 evaluates VLA policies in RoboTwin simulation; tracks closed 2026-05-15; champion team 'zzl' (UESTC). Organisers include GigaAI and RoboTwin author Yao Mu.",
       "level": "verified",
       "sources": [
        "s41"
       ]
      }
     ],
     "short": "3 challenges built on it"
    },
    "derived_benchmarks": {
     "value": [
      "RMBench",
      "ForesightSafety-VLA",
      "WorldArena"
     ],
     "level": "verified",
     "sources": [
      "s42",
      "s43",
      "s44"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "RMBench",
       "display": "2026-03, by the RoboTwin team. Memory-dependent manipulation tasks built on RoboTwin 2.0.",
       "level": "verified",
       "sources": [
        "s42"
       ]
      },
      {
       "value": "ForesightSafety-VLA",
       "display": "2026-06. 66 safety scenarios in RoboTwin across 5 robots.",
       "level": "verified",
       "sources": [
        "s43"
       ]
      },
      {
       "value": "WorldArena",
       "display": "2026-02. Scores world models on RoboTwin 2.0 videos and compares their policy rankings with the RoboTwin simulator (simulator, not real robots).",
       "level": "verified",
       "sources": [
        "s44"
       ]
      }
     ],
     "short": "3 benchmarks built on it"
    },
    "citations": {
     "value": 595,
     "display": "595 (Semantic Scholar; 265 influential)",
     "level": "verified",
     "sources": [
      "s47"
     ],
     "checked": "2026-10-10",
     "short": "595"
    },
    "github_stars": {
     "value": 2950,
     "display": "2,950 stars, 515 forks (RoboTwin-Platform/RoboTwin)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "2,950"
    },
    "dataset_downloads": {
     "value": 105159,
     "display": "105,159 (Hub 'downloads' field), 721,054 all time, 71 likes (TianxingChen/RoboTwin2.0)",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "105,159 on Hugging Face"
    },
    "used_by": {
     "value": "One of the five manipulation benchmarks the 2026 audit calls most widely reported. 24 models on the official board; many more report the 50+500 protocol in papers.",
     "level": "verified",
     "sources": [
      "s30",
      "s11",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list. The audit's data file lists 19 papers claiming a new best on the 50+500 protocol.",
     "items": [
      {
       "value": "X-VLA",
       "display": "2025-10. Reports Easy 70.0, Hard 39.0; training data for RoboTwin not stated.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "Motus",
       "display": "Tsinghua and others, 2025-12. 88.66 clean / 87.02 randomized under the 50+500 protocol it introduced; also scored GO-1, π0.5 and X-VLA.",
       "level": "verified",
       "sources": [
        "s34"
       ]
      },
      {
       "value": "Fast-WAM",
       "display": "Tsinghua IIIS and Galaxea AI, 2026-03. 91.88 / 91.78 under the 50+500 protocol.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      },
      {
       "value": "MotuBrain",
       "display": "ShengShu, 2026-04. 95.8 / 96.1 under the 50+500 protocol.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "GigaBrain-0.7",
       "display": "GigaAI, 2026-08. Official co-train protocol; 66.8 / 67.9.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      }
     ],
     "short": "Widely reported. 24 models are on the official leaderboard."
    },
    "industry_use": {
     "value": [
      "Li Auto",
      "GigaAI",
      "ManiFold AI",
      "ShengShu",
      "Galaxea AI",
      "Hugging Face",
      "D-Robotics",
      "AgiBot"
     ],
     "level": "verified",
     "sources": [
      "s11",
      "s37",
      "s36",
      "s35",
      "s24",
      "s40",
      "s49"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Li Auto, GigaAI, ManiFold AI",
       "display": "Submitted leaderboard entries (ME-Dex-1.0, GigaBrain-0.7, WorldScape Policy 2.0). GigaAI also runs a challenge with a RoboTwin track.",
       "level": "verified",
       "sources": [
        "s11",
        "s41"
       ]
      },
      {
       "value": "ShengShu, Galaxea AI",
       "display": "Report RoboTwin 2.0 results in their model papers (MotuBrain; Fast-WAM).",
       "level": "verified",
       "sources": [
        "s36",
        "s35"
       ]
      },
      {
       "value": "Hugging Face",
       "display": "LeRobot ships a RoboTwin 2.0 environment and evaluation recipe (2026-04).",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "D-Robotics",
       "display": "Co-affiliation and software support; hosted the Challenge Cup 2025 RoboTwin evaluation.",
       "level": "verified",
       "sources": [
        "s5",
        "s40"
       ]
      },
      {
       "value": "AgiBot",
       "display": "GO-1 repository supports RoboTwin simulation evaluation.",
       "level": "verified",
       "sources": [
        "s49"
       ]
      }
     ]
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s11",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Leaderboard updated 2026-10-10; code and data changed in 2026-09; 89 open issues and pull requests."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Two training protocols give very different Hard scores",
     "text": "The official board trains on 50 clean demonstrations per task; its best Hard score is 68.12. Since Motus (2025-12) many papers train one policy on 50 clean plus 500 randomized demonstrations per task, which puts randomized scenes into training; MotuBrain reports 96.1 Hard. The same model names differ widely: Fast-WAM 91.78 Hard in its paper vs FastWAM 1.9 Hard on the board; the audit's data file lists starVLA-α at 88.3 and ABot-M0 at 85.08, while the board shows starVLA 3.16 and Abot-M0 30.36. Both protocols are called 'RoboTwin 2.0' in papers.",
     "level": "verified",
     "sources": [
      "s11",
      "s34",
      "s35",
      "s36",
      "s31"
     ],
     "status": "open",
     "note": "The starVLA-α and ABot-M0 paper values are taken from the audit's data file, not from those papers.",
     "short": "Depending on the training data, Hard scores for the same model range from about 2% to 92%."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same baseline gets different scores in different sources",
     "text": "π0.5: 70.7 / 46.0 on the official board; 42.98 / 43.84 in Motus's run and 82.74 / 76.76 in Fast-WAM's table, both under the 50+500 protocol. X-VLA: 70.0 / 39.0 in its own paper (training data not stated); 68.0 / 20.9 on the board; 72.8 / 72.84 in Motus's run. The audit replaced X-VLA's own number with a protocol-matched one because X-VLA did not train on 50 clean plus 500 randomized demonstrations.",
     "level": "verified",
     "sources": [
      "s11",
      "s34",
      "s35",
      "s33",
      "s30"
     ],
     "status": "open",
     "short": "Different sources report the Hard score of π0.5 as 46.0, 43.84 and 76.76."
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "Seeds did not reproduce the same scene until September 2026",
     "text": "Users found that a dataset seed did not recreate the recorded scene (issue #288). Causes: asset order depended on the file system in 5 tasks, table-clutter bounds grew with each setup call, and Python's random generator, which picks the instruction, was not seeded. Fixes were merged on 2026-09-24 (PRs #490 and #509). A separate open bug (#522, 2026-10-09) shows batch evaluation can score a different set of seeds depending on the number of workers.",
     "level": "verified",
     "sources": [
      "s17",
      "s18",
      "s16"
     ],
     "status": "open",
     "mitigation": {
      "text": "Seed fixes merged 2026-09-24; results produced before then may not be exactly repeatable on another machine.",
      "sources": [
       "s17"
      ]
     },
     "short": "The same seed (the number that sets up a random scene) could produce a different scene. This was fixed in September 2026. A separate bug in batch evaluation is still open."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "Success checks can count brief moments and some disagree with the instructions",
     "text": "An episode ends as a success the first time the task predicate holds, without checking that objects stay put; users report bottles and stacked blocks counted as successes while still moving or falling (issues #493 and #491, open). rotate_qrcode instructions asked to lift and rotate while the check required putting the object down (fixed 2026-09-22); some hanging_mug instructions name the wrong arm (#521, open). LeRobot's docs say open_laptop's check fails during normal policy evaluation; RoboTwin fixed the arm-tag setup for open_laptop, place_object_scale and put_object_cabinet on 2026-08-20 (PR #488), after LeRobot's note.",
     "level": "verified",
     "sources": [
      "s19",
      "s20",
      "s21",
      "s22",
      "s23",
      "s24"
     ],
     "status": "open",
     "short": "A success can be recorded while an object is still falling. Some instructions ask for something different from what the success check tests."
    },
    {
     "id": "i5",
     "type": "protocol-variance",
     "title": "Code and data changed while the leaderboard was running",
     "text": "DP3 evaluation code was fixed on 2025-07-19 and ACT deployment code on 2025-08-25, with the board updated. Several success checks changed in July and August 2025. The official demo_clean training data was refreshed on 2026-09-14 with 'legal-joint replacements', after most co-train entries were listed. Paper v1 reported clean-to-randomized drops of 28.2 (RDT) and 40.3 (π0) points on 13 tasks; v2 reports 20.8 and 30.1 on 50 tasks. Board entries date from 2025-08 to 2026-10 and so span these changes.",
     "level": "verified",
     "sources": [
      "s5",
      "s15",
      "s3",
      "s2",
      "s11"
     ],
     "status": "open",
     "short": "Leaderboard entries were produced with different versions of the code and data."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Image colour channels differ between data versions",
     "text": "The README says older ('legacy') data stores red and blue swapped and was never migrated, so old and new formats can appear in one training run; images must be decoded with decode_image_bit. A user reported the swap in 2026-04 (issue #438); a maintainer replied that images follow RGB order. The demo_clean data and LeRobot archives were refreshed with standard RGB images on 2026-09-14.",
     "level": "verified",
     "sources": [
      "s5",
      "s25",
      "s15"
     ],
     "status": "addressed",
     "short": "Older data stores images with the red and blue channels swapped. The official decoder handles both formats."
    },
    {
     "id": "i7",
     "type": "other",
     "title": "Some leaderboard results cannot be re-checked",
     "text": "The π0 checkpoints behind the board were never released; a maintainer said π0 was not part of the latest re-evaluation (#487). The single-task ACT, DP3 and RDT checkpoints were deleted from the Hugging Face repo on 2026-09-03. Users reported large gaps when reproducing π0 (for example Lift Pot 84% on the board vs 25%, #232, closed as solved) and DP3 (#441, open). Maintainers reproduced four tasks within 4 points of the board in 2025-10 (#215).",
     "level": "verified",
     "sources": [
      "s26",
      "s15",
      "s27",
      "s28",
      "s29"
     ],
     "status": "open",
     "short": "The trained models (checkpoints) behind several leaderboard baselines are not available. Users report gaps when they try to reproduce some results."
    },
    {
     "id": "i8",
     "type": "shortcut",
     "title": "DP3 uses information a real robot would not have",
     "text": "The paper says DP3's strong clean-scene results 'partly stem from perfect point clouds and clean background segmentation in simulation'; its training setup uses precise segmentation of background and tabletop.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "DP3's Easy score relies partly on perfect depth data and perfect separation of objects from the background, which only the simulator provides."
    },
    {
     "id": "i9",
     "type": "other",
     "title": "Bundled 3D models have non-commercial terms despite an MIT label",
     "text": "The data repository is labelled MIT but includes PartNet-Mobility models, whose terms limit use to non-commercial research and education and require users to accept them, and Objaverse objects with per-object licences, some non-commercial.",
     "level": "inferred",
     "sources": [
      "s2",
      "s14",
      "s45",
      "s46"
     ],
     "status": "open",
     "note": "Link between RoboTwin's files and these sources rests on the paper's description of RoboTwin-OD. Not legal advice.",
     "short": "The data is labelled MIT, but some of the bundled 3D models come with non-commercial terms."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "RoboTwin 2.0 tests two-arm skills under visual clutter, and the 2026 audit found no cheap shortcut on it. A high score still says little about real robots. No study has scored the same policies on version 2.0 and on real hardware.",
     "basis": [
      "facts.independent_audit",
      "facts.sim_to_real"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "RoboTwin 2.0 is a useful simulation test. Its link to real-robot results has not been measured."
    },
    {
     "id": "r2",
     "text": "Before comparing numbers, check the training protocol. The official leaderboard uses 50 clean demonstrations per task. The other protocol adds 500 randomized demonstrations per task. Hard scores under the two are not comparable.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "facts.metric_detail"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check which training protocol a score used."
    },
    {
     "id": "r3",
     "text": "Read Easy and Hard separately. Several policies do well in clean scenes and score near zero in randomized ones. On the leaderboard, FastWAM scores 77.8 on Easy and 1.9 on Hard. The leaderboard's average hides this.",
     "basis": [
      "facts.top_score",
      "issues.i1"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Look at the Hard scores as well as the average."
    },
    {
     "id": "r4",
     "text": "Small gaps between leaderboard entries are hard to interpret. There are no error bars, and entries were produced by different teams on different code and data versions. Some entries appear to use 50 rather than 100 trials.",
     "basis": [
      "facts.uncertainty_reported",
      "facts.trials",
      "issues.i5",
      "issues.i3"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Ignore small gaps between leaderboard entries."
    }
   ],
   "searched": [
    {
     "for": "validity (paired sim-versus-real evaluations on RoboTwin 2.0)",
     "where": "RoboTwin 2.0 paper v1 and v2 (real runs test data transfer only), ICML page, README, leaderboard page and data file (its RoboDojo tabs belong to a separate benchmark), CVPR 2025 challenge report (real round used different tasks; no sim-real comparison), WorldEval (RoboTwin 1.0, custom tasks), WorldArena and BWM (world models compared with the RoboTwin simulator, not real robots), 'Toward Visually Realistic Simulation' (2605.06311; its comparison table marks RoboTwin as having no validated sim-to-real correlation), the 2026 audit (no sim-real test). One web search (extended) for RoboTwin 2.0 sim-real correlation studies; the shared web-search budget then ran out, so later papers were not searched again.",
     "date": "2026-10-10"
    },
    {
     "for": "trials and protocol details",
     "where": "Official eval script (scripts/eval_policy_xpolicylab.py), task configs demo_clean.yml and demo_randomized.yml, leaderboard data file (per-task values checked for parity), GigaBrain-0.7 report.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "Repo LICENSE, Hugging Face card (metadata only), Hugging Face file tree (objects.zip, background_texture.zip, embodiments.zip), RoboTwin docs index, PartNet-Mobility terms (sapien-sim Hugging Face gate; sapien.ucsd.edu refused connection), Objaverse card.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "RoboTwin 2.0 (arXiv abstract page, authors, submission history)",
     "url": "https://arxiv.org/abs/2506.18088",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "RoboTwin 2.0 full text v2",
     "url": "https://arxiv.org/html/2506.18088v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "RoboTwin 2.0 full text v1",
     "url": "https://arxiv.org/html/2506.18088v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "RoboTwin 2.0, ICML 2026 proceedings page (PMLR 306:13673-13699)",
     "url": "https://proceedings.mlr.press/v306/chen26j.html",
     "type": "paper",
     "publisher": "PMLR",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "RoboTwin-Platform/RoboTwin README (update log, data notes)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/blob/main/README.md",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: RoboTwin-Platform/RoboTwin (stars, forks, pushed)",
     "url": "https://api.github.com/repos/RoboTwin-Platform/RoboTwin",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "RoboTwin LICENSE (MIT)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/blob/main/LICENSE",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "RoboTwin scripts/requirements.txt (sapien==3.0.0b1, mplib==0.2.1)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/blob/main/scripts/requirements.txt",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "RoboTwin evaluation script scripts/eval_policy_xpolicylab.py (seeds, expert check, test_num)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/blob/main/scripts/eval_policy_xpolicylab.py",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "RoboTwin task configs demo_clean.yml and demo_randomized.yml",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/tree/main/env_cfg/task_config",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "RoboTwin 2.0 leaderboard data file (24 entries, per-task Easy and Hard, news)",
     "url": "https://robotwin-platform.github.io/data/robotwin_leaderboard.json",
     "type": "leaderboard",
     "publisher": "RoboTwin Platform",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "RoboTwin 2.0 leaderboard page (setting, listing policy)",
     "url": "https://robotwin-platform.github.io/leaderboard",
     "type": "leaderboard",
     "publisher": "RoboTwin Platform",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "TianxingChen/RoboTwin2.0 dataset card and Hub API (licence, gated, downloads)",
     "url": "https://huggingface.co/api/datasets/TianxingChen/RoboTwin2.0?expand[]=downloads&expand[]=downloadsAllTime&expand[]=likes&expand[]=gated&expand[]=lastModified&expand[]=cardData",
     "type": "repo",
     "publisher": "Hugging Face / RoboTwin team",
     "date": "2026-09-22",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "TianxingChen/RoboTwin2.0 file tree (per-task archives, assets, co-train checkpoints)",
     "url": "https://huggingface.co/datasets/TianxingChen/RoboTwin2.0/tree/main",
     "type": "repo",
     "publisher": "RoboTwin team",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "TianxingChen/RoboTwin2.0 commit history (data refresh 2026-09-14, checkpoint deletions 2026-09-03)",
     "url": "https://huggingface.co/api/datasets/TianxingChen/RoboTwin2.0/commits/main",
     "type": "repo",
     "publisher": "RoboTwin team",
     "date": "2026-09-22",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "RoboTwin issue #522: batch evaluation can select different seed sets (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/522",
     "type": "repo",
     "publisher": "RoboTwin Platform (user report)",
     "date": "2026-10-09",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "RoboTwin PR #509 'make a given seed reproduce the same episode' (merged 2026-09-24; refs issue #288, PR #490)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/pull/509",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09-24",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "RoboTwin issue #505: same seed can produce different instructions (closed)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/505",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09-15",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "RoboTwin issue #493: success may be triggered by transient states (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/493",
     "type": "repo",
     "publisher": "RoboTwin Platform (user report)",
     "date": "2026-08-25",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "RoboTwin issue #491: post-settle stability for block-stack success (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/491",
     "type": "repo",
     "publisher": "RoboTwin Platform (user report)",
     "date": "2026-08-23",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "RoboTwin issue #506: rotate_qrcode instruction vs success check (closed 2026-09-22)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/506",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09-15",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "RoboTwin issue #521: hanging_mug instructions assign the wrong arm (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/521",
     "type": "repo",
     "publisher": "RoboTwin Platform (user report)",
     "date": "2026-10-09",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "RoboTwin PR #488: pre-set task arm tags (merged 2026-08-20) and envs/open_laptop.py",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/pull/488",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-08-20",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "LeRobot documentation: RoboTwin 2.0",
     "url": "https://github.com/huggingface/lerobot/blob/main/docs/source/robotwin.mdx",
     "type": "repo",
     "publisher": "Hugging Face",
     "date": "2026-04-20",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "RoboTwin issue #438: possible RGB/BGR inversion in HDF5 images (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/438",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-04-14",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "RoboTwin issue #487: release the π0 leaderboard checkpoints (maintainer reply)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/487",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-09-03",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "RoboTwin issue #232: reproducing π0 leaderboard results",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/232",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "RoboTwin issue #441: DP3 success rates vs the paper (open)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/441",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "RoboTwin issue #215: data generated differently between versions (maintainer reproduction)",
     "url": "https://github.com/RoboTwin-Platform/RoboTwin/issues/215",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (Table 1, Section 4, Appendix A.1)",
     "url": "https://arxiv.org/html/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "Audit data file: stat_sig_robotwin2_hard_randomized_cutoff_data.csv (19 comparisons)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/analysis/paper_inputs/stat_sig_robotwin2_hard_randomized_cutoff_data.csv",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Audit Figure 3: statistical significance pies (RoboTwin 2.0, n=19)",
     "url": "https://arxiv.org/html/2606.04233v1/statistical_significance_pies.svg",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "X-VLA: Soft-Prompted Transformer as Scalable Cross-Embodiment VLA Model (Table 2, Table 16)",
     "url": "https://arxiv.org/html/2510.10274",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Motus: A Unified Latent Action World Model (RoboTwin 2.0 protocol, Table 13)",
     "url": "https://arxiv.org/html/2512.13030v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Fast-WAM: Do World Action Models Need Test-time Future Imagination? (Table 1)",
     "url": "https://arxiv.org/html/2603.16666",
     "type": "paper",
     "publisher": "arXiv (Tsinghua IIIS, Galaxea AI)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "MotuBrain: An Advanced World Action Model for Robot Control (Table 3)",
     "url": "https://arxiv.org/html/2604.27792",
     "type": "paper",
     "publisher": "arXiv (MotuBrain Team, ShengShu)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "GigaBrain-0.7 technical report (Table 9, Appendix A.1)",
     "url": "https://arxiv.org/html/2608.15875",
     "type": "paper",
     "publisher": "arXiv (GigaAI)",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "WorldEval: World Model as Real-World Robot Policies Evaluator (Section 4.2, real-to-sim comparison on RoboTwin 1.0)",
     "url": "https://arxiv.org/html/2505.19017",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-05",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "Benchmarking Generalizable Bimanual Manipulation: RoboTwin Dual-Arm Collaboration Challenge at CVPR 2025 MEIS Workshop",
     "url": "https://arxiv.org/html/2506.23351",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "挑战杯 2025 '人工智能+' 挑战赛题目四：端侧可部署的双臂操作算法设计 (problem statement PDF)",
     "url": "https://2025.tiaozhanbei.net/media/ckeditor_uploads/49/2025/05/14/4.%E3%80%90%E9%A2%98%E7%9B%AE%E5%9B%9B%E3%80%91%E7%AB%AF%E4%BE%A7%E5%8F%AF%E9%83%A8%E7%BD%B2%E7%9A%84%E5%8F%8C%E8%87%82%E6%93%8D%E4%BD%9C%E7%AE%97%E6%B3%95%E8%AE%BE%E8%AE%A1.pdf",
     "type": "site",
     "publisher": "挑战杯 organising committee",
     "date": "2025-05-14",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "GigaBrain Challenge 2026 (CVPR 2026 workshop) page",
     "url": "https://gigaai-research.github.io/GigaBrain-Challenge-2026/",
     "type": "site",
     "publisher": "GigaAI and co-organisers",
     "date": "2026",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "RMBench repository and arXiv 2603.01229",
     "url": "https://github.com/RoboTwin-Platform/RMBench",
     "type": "repo",
     "publisher": "RoboTwin Platform",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "ForesightSafety-VLA: A Unified Diagnostic Safety Benchmark for VLA Models (abstract)",
     "url": "https://arxiv.org/abs/2606.27079",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "WorldArena: A Unified Benchmark for Evaluating Perception and Functional Utility of Embodied World Models",
     "url": "https://arxiv.org/html/2602.08971",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "PartNet-Mobility terms of use (sapien-sim/PartNetMobility gate text)",
     "url": "https://huggingface.co/datasets/sapien-sim/PartNetMobility",
     "type": "repo",
     "publisher": "UC San Diego, Stanford, Simon Fraser University",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "allenai/objaverse dataset card (licence section)",
     "url": "https://huggingface.co/datasets/allenai/objaverse",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "Semantic Scholar API record for arXiv:2506.18088",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2506.18088?fields=title,citationCount,influentialCitationCount,venue,publicationDate,externalIds",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "Toward Visually Realistic Simulation: A Benchmark for Evaluating Robot Manipulation in Simulation (comparison table)",
     "url": "https://arxiv.org/html/2605.06311v1",
     "type": "secondary",
     "publisher": "arXiv (other authors)",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "OpenDriveLab/AgiBot-World README (GO-1 examples incl. RoboTwin)",
     "url": "https://github.com/OpenDriveLab/AgiBot-World/blob/main/README.md",
     "type": "repo",
     "publisher": "OpenDriveLab / AgiBot",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "Semantic Scholar record for RoboTwin 1.0 (arXiv:2504.13059, CVPR 2025)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2504.13059?fields=title,citationCount,influentialCitationCount,venue,publicationDate,externalIds",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    },
    {
     "date": "2026-10-10",
     "change": "Expanded to a full entry from primary sources: paper v1 and v2, ICML page, repo code and configs, leaderboard data, Hugging Face data history, GitHub issues, the 2026 audit and its data file, model papers reporting results, challenge documents and asset licence terms. Added the two-protocol problem, reproducibility and success-check issues, top scores, challenges and derived benchmarks. Corrected: LeRobot's open_laptop note is outdated (fixed upstream 2026-08-20). WorldEval's r 0.411 is kept out of validity because it used RoboTwin 1.0."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well a policy will do on a real robot.",
     "sub": "No study has scored the same policies on RoboTwin 2.0 and on real robots.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How a score compares with scores in other papers.",
     "sub": "Papers use two different training protocols, and they give very different Hard scores.",
     "basis": [
      "issues.i1",
      "issues.i2"
     ]
    },
    {
     "id": "l3",
     "text": "How well a policy works on other robots.",
     "sub": "All scores use one simulated robot, the Aloha-AgileX.",
     "basis": [
      "facts.robots",
      "facts.embodiment"
     ]
    },
    {
     "id": "l4",
     "text": "Whether the task stays done after a success is recorded.",
     "sub": "A success is recorded the first moment the goal is met, even if an object is still moving or falling.",
     "basis": [
      "issues.i4"
     ]
    }
   ],
   "validity": []
  },
  {
   "id": "roboworld",
   "name": "RoboWorld",
   "aliases": [
    "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation",
    "Step Forcing (training method)"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "World-model evaluation built for robotics: scores robot policies by closed-loop rollouts in a learned video simulator; explicitly in scope.",
   "summary": {
    "text": "Runs robot policies inside a learned video world model trained on DROID; a VLM judge scores task progress.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "KAIST; Config",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Affiliations in arXiv v4 and project page (1 KAIST, 2 Config). What kind of organisation 'Config' is was not checked."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All authors KAIST; several also Config. Config's type unknown."
    },
    "region": {
     "value": "asia-other",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "KAIST is in South Korea."
    },
    "first_release": {
     "value": "2026-07 (arXiv v1 2026-07-01)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "arXiv v4 2026-07-15 (v2 07-13, v3 07-14)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "What changed between versions is not stated (file sizes nearly identical)."
    },
    "version": {
     "value": "arXiv v4; code 'coming soon'",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Project page shows 'Code coming soon'; no repo linked from paper or page."
    },
    "published_at": {
     "value": "ICML 2026 F2S Workshop on Long-Horizon Video Generation",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Stated on project page; arXiv comments give only the project link."
    },
    "venue": {
     "value": "world-model",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Closed-loop: policy acts on generated frames; world model predicts next frames from actions."
    },
    "robots": {
     "value": "DROID setup (Franka Panda; two external views + one wrist view, tiled 2x2)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Conditioned on end-effector Cartesian position; adapter for joint-velocity policies."
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "world-modeling"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Scores manipulation policies; also reports video quality (SSIM, LPIPS, FVD) of the world model itself."
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Initial frames come from RoboArena episodes (many sites); plus 8 synthetic environments made with an image editor."
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "8 policies; 4,186 rollouts; 30 s per rollout; about 100 H100 GPU-hours for the full RoboArena replication",
       "level": "verified",
       "sources": [
        "s1"
       ],
       "note": "Policies = those open-sourced as of RoboArena data dump 2026-02-03."
      },
      {
       "value": "Synthetic: 175 initial observations -> 746 valid initial conditions across 8 environments",
       "level": "verified",
       "sources": [
        "s1"
       ]
      }
     ]
    },
    "scoring": {
     "value": [
      "auto-judge",
      "progress"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "GPT-4o judge with a 0-5 task-progress rubric scored from fixed external views (wrist view treated as less reliable)."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on page or paper."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Code not released ('Code coming soon'); looked at project page and arXiv links."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No new dataset released; uses DROID (training) and RoboArena data dump (MIT, see RoboArena record). Paper text is CC BY 4.0."
    },
    "sim_to_real": {
     "value": "correlated",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "8 policies; Pearson r 0.989 and Spearman rho 0.970 between RoboWorld scores (GPT-4o judge, 0-5 progress rubric) and the RoboArena real-world leaderboard (snapshot 2026-02-26). Binary-success scoring gives rho 0.922. Synthetic 'extreme' environments still r 0.970 vs RoboArena. Measured by the RoboWorld authors against real data collected by a different group (RoboArena); no third-party replication found."
    },
    "citations": {
     "value": 3,
     "display": "3 (Semantic Scholar; 1 influential)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "2026-10-10."
    },
    "status": {
     "value": "active",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Paper revised four times in July 2026; code promised."
    },
    "kind": {
     "value": "platform",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Stated limitations",
     "text": "Long-horizon, contact-rich manipulation with object consistency remains hard for video world models. Correlation rests on 8 policies and one VLM judge (GPT-4o).",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "status": "open"
    }
   ],
   "readings": [],
   "sources": {
    "s1": {
     "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation (full text)",
     "url": "https://arxiv.org/html/2607.01060v4",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-07"
    },
    "s2": {
     "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation",
     "url": "https://byeongguks.github.io/RoboWorld/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation",
     "url": "https://arxiv.org/abs/2607.01060",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-07"
    },
    "s4": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/batch",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "sima-2-evaluation",
   "name": "SIMA 2 evaluation",
   "full_name": "SIMA 2 evaluation (SIMA Evaluation Suite 2.0)",
   "aliases": [
    "SIMA Evaluation Suite 2.0",
    "SIMA 2"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Scores an embodied agent acting in 3D virtual worlds, but the bodies are game avatars, not robots. The scope rule lists SIMA 2 as borderline. The suite is also closed: no one outside Google DeepMind can re-run it.",
   "summary": {
    "text": "Google DeepMind's internal test of a Gemini-based game agent: task success in 3D video games, compared with human players.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Google DeepMind (SIMA Team)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Title block: 'SIMA Team, Google DeepMind'. arXiv lists Adrian Bolton and 64 other authors."
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Single corporate research lab."
    },
    "region": {
     "value": "multi",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Google DeepMind's careers page lists offices in London, Bay Area, Bangalore, Cambridge (US), Montreal, New York City, Paris, Tokyo, Toronto and Zurich. No headquarters is named on the pages checked, so a single region is not assigned."
    },
    "first_release": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "2025-11 (announcement blog, 2025-11-13); paper arXiv v1 2025-12-04",
       "level": "verified",
       "sources": [
        "s4"
       ],
       "note": "Blog dated 13 Nov 2025; arXiv 2512.04797 v1 submitted 4 Dec 2025 (https://arxiv.org/abs/2512.04797)."
      },
      {
       "value": "Evaluation lineage: SIMA 1 paper, arXiv 2404.10179 v1 2024-03-13",
       "level": "verified",
       "sources": [
        "s5"
       ],
       "note": "SIMA 2 paper says task success is measured 'as in SIMA Team et al. (2024)', i.e. the SIMA 1 evaluation design."
      }
     ]
    },
    "latest_update": {
     "value": "2025-12: arXiv 2512.04797 v1 (no later version); tech report PDF dated 2025-12-05",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "arXiv lists only v1. The PDF linked from the blog carries the date 2025-12-05."
    },
    "version": {
     "value": "SIMA Evaluation Suite 2.0 (as described in arXiv 2512.04797v1)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Section 3.4 names 'SIMA Evaluation Suite 2.0'. Changes from SIMA 1: more domains, more tasks ('often by an order of magnitude' for programmatic evals), success text must persist several seconds, limits on actions after completion, sequential chains must be fully completed."
    },
    "venue": {
     "value": "sim",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Closed-loop keyboard-and-mouse control inside game engines and research environments. These are game engines, not physics simulators built for robotics (see taxonomy_friction)."
    },
    "capability": {
     "value": [
      "instruction-following",
      "navigation",
      "long-horizon"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Agent receives text instructions and acts. Skill categories reported: interaction, navigation, menu use, tool use, construction, object management, resource gathering, combat (Fig. 7, Appendix A). Sequential task chains are part of Suite 2.0."
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "game-world"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Training: Construction Lab, Playhouse, WorldLab (research environments); Goat Simulator 3, Hydroneer, No Man's Sky, Satisfactory, Space Engineers, Valheim, Wobbly Life (commercial). Held-out: ASKA, MineDojo (Minecraft); The Gunk and Genie 3 worlds only qualitatively."
    },
    "scale": {
     "value": "10 training domains (3 research environments + 7 commercial games); held-out quantitative evaluation on ASKA and a 50-task MineDojo subset (15 random seeds per task). Total number of evaluation tasks not stated.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Blog lists more partner games (also Eco, The Gunk, SteamWorld Build, Road 96, Teardown) than the paper's 7 commercial training games; the paper's list is used for the evaluation."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Organiser-run, self-reported by Google DeepMind. Human baselines are shown with and without the agent's time limit. Uncertainty statistics are not described in the text. Values are the bar labels printed in Fig. 6 of the PDF. Bar-to-series mapping read from bar order; consistent with the text's 'SIMA 2 effectively doubles the average success rate of SIMA 1'. Baseline Gemini models without SIMA training: Flash-Lite 3.2%, Pro 7.0% over the 10 training domains (text, https://arxiv.org/html/2512.04797). Human figures are for 'a representative subset of tasks'."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Results appear only in the paper and blog. No public leaderboard found."
    },
    "access": {
     "value": "closed",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "SIMA 2 was released as a 'limited research preview' to a small cohort of academics and game developers. Games used under licensed agreements with their developers."
    },
    "license_code": {
     "value": "not released",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "No evaluation code, task list or repository mentioned in the paper or blog. arXiv's CC BY 4.0 licence covers the paper text only."
    },
    "license_data": {
     "value": "not released",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "No evaluation tasks or data released. Commercial games are third-party products."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No physical robot in the loop and no transfer study found (paper, blog). Real-world validity in the robotics sense does not apply directly to a game agent."
    },
    "kind": {
     "value": "study",
     "level": "inferred",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "SIMA 2: A Generalist Embodied Agent for Virtual Worlds (full text)",
     "url": "https://arxiv.org/html/2512.04797",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s2": {
     "title": "https://storage.googleapis.com/deepmind-media/DeepMind.com/Blog/sima-2-an-agent-that-plays-reasons-and-learns-with-you-in-virtual-3d-worlds/SIMA_Tech_Report_2025.pdf",
     "url": "https://storage.googleapis.com/deepmind-media/DeepMind.com/Blog/sima-2-an-agent-that-plays-reasons-and-learns-with-you-in-virtual-3d-worlds/SIMA_Tech_Report_2025.pdf",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Careers at Google DeepMind — Google DeepMind",
     "url": "https://deepmind.google/careers/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "SIMA 2: A Gemini-Powered AI Agent for 3D Virtual Worlds — Google DeepMind",
     "url": "https://deepmind.google/blog/sima-2-an-agent-that-plays-reasons-and-learns-with-you-in-virtual-3d-worlds/",
     "type": "blog",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "Scaling Instructable Agents Across Many Simulated Worlds",
     "url": "https://arxiv.org/abs/2404.10179",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-03"
    },
    "s6": {
     "title": "SIMA 2: A Generalist Embodied Agent for Virtual Worlds",
     "url": "https://arxiv.org/abs/2512.04797",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-12"
    },
    "s7": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2512.04797",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "simplerenv",
   "name": "SimplerEnv",
   "full_name": "Evaluating Real-World Robot Manipulation Policies in Simulation",
   "aliases": [
    "SIMPLER",
    "Simpler-Bridge / Simpler-WidowX",
    "Simpler-Fractal / Simpler-Google",
    "SimplerEnv-ManiSkill3 (GPU port of the WidowX tasks)"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Simulation benchmark that scores real-data-trained manipulation policies by closed-loop success rate in simulated copies of real setups.",
   "summary": {
    "text": "SimplerEnv (SIMPLER) is a set of simulated copies of two real robot test setups, the Google Robot and the WidowX arm with BridgeData V2 tasks, built so that policies trained on real robot data can be scored in simulation. Its authors found strong agreement with real-robot results for 2023-2024 policies; later checks with newer policies and mostly new tasks found weaker agreement.",
    "short": "SimplerEnv is a set of simulated copies of two real robot test setups, used to score robot policies (the models that control a robot) in simulation. Its scores agreed with real-robot results for policies from 2023 and 2024, and later checks with newer policies found weaker agreement.",
    "sources": [
     "s2",
     "s20",
     "s21"
    ]
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents a suite of simulated evaluation environments with fixed tasks, trial grids and success checks; papers report success rates on it as a benchmark. Classification by the Atlas."
    },
    "kind_secondary": {
     "value": [
      "platform"
     ],
     "display": "Also a workflow and code for building new real-to-sim evaluation environments",
     "level": "verified",
     "sources": [
      "s2",
      "s7",
      "s46"
     ],
     "checked": "2026-10-10",
     "note": "The paper and the repo guide describe how to add robots, objects and scenes. The taxonomy has no value for a method toolkit, so 'platform' is the closest fit."
    },
    "version": {
     "value": "0.0.1",
     "display": "Package simpler_env 0.0.1 with the submodule mani_skill2_real2sim 0.5.3 (pins SAPIEN 2.2.2). No tags or releases. Two branches: main (CPU, ManiSkill2) and maniskill3 (GPU versions of the 4 WidowX tasks).",
     "level": "verified",
     "sources": [
      "s52",
      "s9",
      "s10",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "setup.py files read 2026-10-10. The GitHub tags and releases lists are empty. arXiv has only v1 (2024-05-09); the PMLR camera-ready text differs from it (validity.v4, issues.i9).",
     "items": [
      {
       "value": "main branch",
       "display": "SAPIEN 2.2.2 and ManiSkill2 on CPU. Used for all of the paper's results.",
       "level": "verified",
       "sources": [
        "s7",
        "s9"
       ]
      },
      {
       "value": "maniskill3 branch and ManiSkill3 digital twins",
       "display": "GPU-parallel versions of the 4 WidowX tasks inside ManiSkill3. The ManiSkill docs say up to 60x faster than real-world evaluation and 10x faster than CPU simulation. The branch README says reproducing the original results needs the main branch.",
       "level": "verified",
       "sources": [
        "s11",
        "s19",
        "s42"
       ],
       "note": "Google Robot tasks are not ported. On 2025-10-09 the lead author wrote that the Google Robot controller migration was still pending."
      },
      {
       "value": "Visual Matching and Variant Aggregation",
       "display": "Two official setups. Visual Matching overlays real background images and tunes object and arm textures. Variant Aggregation averages over randomised backgrounds, lighting, distractors and table textures (Google Robot only).",
       "level": "verified",
       "sources": [
        "s2",
        "s7"
       ]
      }
     ],
     "short": "0.0.1. There are no tagged releases."
    },
    "publishers": {
     "value": [
      "UC San Diego",
      "Stanford University",
      "UC Berkeley",
      "Google DeepMind"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "16 authors. Equal contribution: Xuanlin Li, Kyle Hsu, Jiayuan Gu. Co-advising: Hao Su, Quan Vuong, Ted Xiao.",
     "items": [
      {
       "value": "UC San Diego",
       "display": "Hao Su's lab; lead authors and the SAPIEN/ManiSkill code base.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Stanford University",
       "display": "Kyle Hsu, Chelsea Finn, Jiajun Wu and others.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "UC Berkeley",
       "display": "Sergey Levine's lab; WidowX real-world evaluations.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Google DeepMind",
       "display": "RT-series policies; the real Google Robot evaluations were run by Google staff, per the lead author in issue #77.",
       "level": "verified",
       "sources": [
        "s2",
        "s50"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2",
      "s50"
     ],
     "checked": "2026-10-10",
     "note": "Three universities and Google DeepMind. Could also be read as academic plus frontier lab."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All four organisations are in the United States."
    },
    "first_release": {
     "value": "2024-05",
     "display": "arXiv v1 on 2024-05-09. Published at CoRL 2024 (PMLR volume 270, pages 3705-3728; volume dated 2025-01-12).",
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub repository was created on 2024-03-23, before the paper.",
     "short": "May 2024, at CoRL 2024"
    },
    "latest_update": {
     "value": "2026-09",
     "display": "2026-09-30: a dependency pin (transformers<5) merged on main. Last change to tasks or environments: October 2024 (GPU port of the WidowX tasks).",
     "level": "verified",
     "sources": [
      "s10",
      "s11",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Other commits on main since 2025: 2025-02-25, 2025-03-28 and 2025-12-20 (README CUDA note). The maniskill3 branch was last changed 2024-10-30; ManiSkill2_real2sim was last pushed 2024-10-19.",
     "short": "September 2026. It was only a fix to a software dependency."
    },
    "status": {
     "value": "maintained",
     "display": "Occasional dependency fixes; no new tasks since 2024. Use is very active (facts.used_by).",
     "level": "inferred",
     "sources": [
      "s10",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "33 open issues on 2026-10-10. The lead author still answers issues (latest reply 2026-09-25).",
     "short": "Only fixes since 2024. Use is very active."
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "Rigid-object picking, moving, drawer and placing tasks given as language instructions. In Move Near and the drawer tasks the instruction names which object or drawer to use. The paper lists rigid objects only as a limitation."
    },
    "generalisation": {
     "value": [
      "object-pose",
      "visual"
     ],
     "display": "Object and robot start positions vary over fixed grids. The Variant Aggregation setup also varies background, lighting, distractors and table texture.",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The tasks were chosen to be representative of tasks in the training datasets (paper Section V), so test tasks match training tasks. Visual Matching keeps visuals fixed to the real setup.",
     "short": "Start positions change. One setup also changes the visuals."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Real-to-sim: simulated copies of specific real evaluation setups."
    },
    "simulator": {
     "value": "SAPIEN 2.2 / ManiSkill2 (main branch); ManiSkill3 for the GPU WidowX port",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s11",
      "s19"
     ],
     "checked": "2026-10-10",
     "note": "Paper Section V builds on SAPIEN; Section VI-D reproduces the Google Robot Variant Aggregation setup in Isaac Sim (no Isaac Sim code found in the repos). Control runs at 3 Hz for the Google Robot and 5 Hz for WidowX, simulation at about 500 Hz (README). One environment renders 3.5k simulation steps per second on an RTX 4090 at 640 x 512 (paper).",
     "short": "SAPIEN and ManiSkill2"
    },
    "embodiment": {
     "value": [
      "single-arm",
      "mobile-manipulator"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The Google Robot is a mobile manipulator placed at fixed floor positions; its base does not move during a task. The WidowX-250 6-DoF is a fixed single arm."
    },
    "robots": {
     "value": "Google Robot (RT-1 robot); WidowX-250 6-DoF (BridgeData V2 setup)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The Google Robot uses a head-mounted camera and the WidowX setup a fixed Logitech C920 third-person camera. Neither setup has a wrist camera."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Tabletop and cabinet-drawer tasks. Real background images are pasted behind the simulated objects ('green screening'). The WidowX eggplant task uses a toy sink."
    },
    "tasks": {
     "value": 8,
     "display": "8 task families: 4 for the Google Robot (pick coke can, move near, open/close drawer, open drawer and place apple) and 4 for WidowX (spoon on towel, carrot on plate, stack blocks, eggplant in basket). The code registers 10 environments with sub-variants.",
     "level": "verified",
     "sources": [
      "s3",
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "PMLR abstract: eight task families. The README lists 10 prepackaged environments, including a pick-random-object environment that the paper's validation does not use.",
     "short": "8 task families on 2 robots"
    },
    "objects": {
     "value": 18,
     "display": "18 distinct objects in the 8 task families (our count)",
     "level": "inferred",
     "sources": [
      "s2",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Counted from the task descriptions in Appendix B: 9 for the Google Robot (coke can, 7 other Move Near objects, cabinet) and 9 for WidowX (spoon, towel, carrot, plate, two blocks, eggplant, basket, sink). The asset folder holds 52 model entries, many of them texture variants."
    },
    "demonstrations": {
     "value": "none shipped",
     "display": "No demonstrations ship with SimplerEnv. Policies are trained on real robot data: the RT-1 (Fractal) data for the Google Robot and BridgeData V2 for WidowX.",
     "level": "verified",
     "sources": [
      "s2",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "System identification used 20 trajectories from these datasets (camera-ready Section 4.1). The benchmark does not restrict training data, so users can also train on simulated demonstrations (issues.i3, issues.i8)."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Binary success per episode. WidowX results also report partial 'grasp' success. MMRV and Pearson r are used to validate the simulator, not to score policies."
    },
    "metric_detail": {
     "value": "success rate per task, averaged per robot setup",
     "display": "Papers report three headline numbers: the WidowX average over 4 tasks (Visual Matching), the Google Robot Visual Matching average and the Google Robot Variant Aggregation average. Google Robot averages cover 3 task groups in some papers and 4 in others.",
     "level": "inferred",
     "sources": [
      "s2",
      "s26",
      "s27",
      "s28",
      "s33"
     ],
     "checked": "2026-10-10",
     "note": "The SIMPLER paper itself reports per-task results and correlation metrics, not one average. In the paper, Google Robot Visual Matching results are averaged over four tuned arm colours and Octo results over three seeds (Section VI-A).",
     "short": "Success rate, reported as three headline averages"
    },
    "trials": {
     "value": "24 to 300 per task, varies by paper",
     "level": "verified",
     "sources": [
      "s2",
      "s39",
      "s40",
      "s25",
      "s31",
      "s36",
      "s14"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Real-world grids (paper)",
       "display": "Google Robot: pick coke can 75, move near 60, open/close drawer 54, drawer + apple 27. WidowX: 24 per task. Simulation multiplies these by variants, arm colours and seeds.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "Appendix B. For put-eggplant-in-basket the real success values (0.233, 0.433, 0.033) are multiples of 1/30, not 1/24 (our arithmetic)."
      },
      {
       "value": "Lead author's advice",
       "display": "Run at least 75 trials per Bridge task (repeat the 24-trial grid 3 times); 25 trials is too noisy.",
       "level": "verified",
       "sources": [
        "s39",
        "s40"
       ],
       "note": "Replies in issues #108 (2025-06-24) and #125 (2025-12-19)."
      },
      {
       "value": "CogACT",
       "display": "Each WidowX task repeated 5 times.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "FASTer",
       "display": "120 trials per WidowX task instead of the default 24.",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "STARE-VLA",
       "display": "300 evaluation episodes, averaged over 3 seeds; best checkpoint chosen by the same evaluation.",
       "level": "verified",
       "sources": [
        "s36"
       ]
      },
      {
       "value": "2026 audit reruns",
       "display": "Each of the 24 grid start states repeated 12 times (288 episodes per task) with the standard step limit (60 steps for stacking).",
       "level": "verified",
       "sources": [
        "s14"
       ]
      }
     ],
     "short": "24 to 300 per task, depending on the paper"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s14",
      "s25",
      "s26",
      "s27",
      "s28",
      "s29",
      "s32",
      "s33",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "In the 14 SimplerEnv papers we opened, SimplerEnv success rates are single numbers without intervals (some give intervals for LIBERO only). The SIMPLER paper runs Kruskal-Wallis tests per policy but gives no intervals. The 2026 audit found 62 of 122 claimed WidowX gains cannot be tested from published averages."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s5",
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "No organiser runs submissions. Each paper runs its own evaluation."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "inferred",
     "sources": [
      "s5",
      "s7",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "No leaderboard on the project site or in the README. The closest public compilation is the 2026 audit's tracker CSV of reported numbers."
    },
    "top_score": {
     "value": 97.9,
     "display": "97.9% WidowX average (CORAL, March 2026). Highest WidowX average without RL in the simulator that we found. Rows below use different protocols and are not directly comparable.",
     "level": "inferred",
     "sources": [
      "s34",
      "s14",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Each score was read in its paper. 'Highest' is our judgement from the 2026 audit's tracker (snapshot 2026-05-21) and the papers we opened. All rows are WidowX (Bridge) Visual Matching success rates (%) over 4 tasks, as each paper reports them; trial counts and step limits differ. The audit's reruns of five of these policies scored lower (issues.i4).",
     "items": [
      {
       "value": 51.3,
       "display": "CogACT, 2024-11: 71.7 / 50.8 / 15.0 / 67.5",
       "level": "verified",
       "sources": [
        "s25"
       ],
       "note": "Spoon / carrot / stack / eggplant. Google Robot: Visual Matching 74.8, Variant Aggregation 61.3 (4-task averages).",
       "data": {
        "model": "CogACT",
        "date": "2024-11",
        "avg": 51.3,
        "rl": false
       }
      },
      {
       "value": 42.7,
       "display": "SpatialVLA (fine-tuned), 2025-01: 16.7 / 25.0 / 29.2 / 100.0",
       "level": "verified",
       "sources": [
        "s26"
       ],
       "note": "34.4 without Bridge fine-tuning. Google Robot: 75.1 and 70.7 are 3-task averages.",
       "data": {
        "model": "SpatialVLA",
        "date": "2025-01",
        "avg": 42.7,
        "rl": false
       }
      },
      {
       "value": 72.7,
       "display": "EO-1, 2025-08: 63.6 / 54.5 / 81.8 / 90.9",
       "level": "verified",
       "sources": [
        "s27"
       ],
       "note": "Google Robot: 76.5 and 63.0 (4-task averages).",
       "data": {
        "model": "EO-1",
        "date": "2025-08",
        "avg": 72.7,
        "rl": false
       }
      },
      {
       "value": 71.7,
       "display": "InternVLA-M1, 2025-10: 87.5 / 67.9 / 31.3 / 100.0",
       "level": "verified",
       "sources": [
        "s28"
       ],
       "note": "Google Robot: 80.7 and 76.0 (4-task averages).",
       "data": {
        "model": "InternVLA-M1",
        "date": "2025-10",
        "avg": 71.7,
        "rl": false
       }
      },
      {
       "value": 95.8,
       "display": "X-VLA (0.9B), 2025-10: 100 / 91.7 / 95.8 / 95.8",
       "level": "verified",
       "sources": [
        "s29"
       ],
       "note": "Google Robot: Visual Matching 80.4, Variant Aggregation 75.7. Bridge fine-tuning with a two-step adaptation.",
       "data": {
        "model": "X-VLA (0.9B)",
        "date": "2025-10",
        "avg": 95.8,
        "rl": false
       }
      },
      {
       "value": 84.4,
       "display": "Dexbotic DB-MemVLA, 2025-10: 100.0 / 66.7 / 70.8 / 100.0",
       "level": "verified",
       "sources": [
        "s30"
       ],
       "data": {
        "model": "Dexbotic DB-MemVLA",
        "date": "2025-10",
        "avg": 84.4,
        "rl": false
       }
      },
      {
       "value": 87.9,
       "display": "FASTer, 2025-12: 91.7 / 93.3 / 67.5 / 99.2",
       "level": "verified",
       "sources": [
        "s31"
       ],
       "note": "Trained from scratch on Bridge; 120 trials per task.",
       "data": {
        "model": "FASTer",
        "date": "2025-12",
        "avg": 87.9,
        "rl": false
       }
      },
      {
       "value": 79.2,
       "display": "Xiaomi-Robotics-0, 2026-02: 95.8 / 62.5 / 75.0 / 83.3",
       "level": "verified",
       "sources": [
        "s32"
       ],
       "note": "Google Robot: 85.5 and 74.7 (4-task averages).",
       "data": {
        "model": "Xiaomi-Robotics-0",
        "date": "2026-02",
        "avg": 79.2,
        "rl": false
       }
      },
      {
       "value": 95.8,
       "display": "SimVLA, 2026-02: 100 / 91.7 / 91.7 / 100",
       "level": "verified",
       "sources": [
        "s33"
       ],
       "note": "Google Robot Variant Aggregation 76.1 (3-task average). CORAL, by the same team, lists SimVLA's per-task scores as 100 / 100 / 91.7 / 91.7.",
       "data": {
        "model": "SimVLA",
        "date": "2026-02",
        "avg": 95.8,
        "rl": false
       }
      },
      {
       "value": 97.9,
       "display": "CORAL on SimVLA, 2026-03: 100.0 / 100.0 / 95.8 / 95.8",
       "level": "verified",
       "sources": [
        "s34"
       ],
       "note": "Trains one LoRA expert per task and picks it from the instruction at run time. Google Robot Variant Aggregation 84.9 (3-task average).",
       "data": {
        "model": "CORAL (SimVLA)",
        "date": "2026-03",
        "avg": 97.9,
        "rl": false
       }
      },
      {
       "value": 86.7,
       "display": "piRL on pi0, 2025-10: 91.6 / 95.7 / 63.0 / 96.7",
       "level": "verified",
       "sources": [
        "s35"
       ],
       "note": "Supervised training on 144 demonstrations per task, then RL inside the simulator (67.2 before RL).",
       "data": {
        "model": "piRL on pi0",
        "date": "2025-10",
        "avg": 86.7,
        "rl": true
       }
      },
      {
       "value": 98,
       "display": "STARE-VLA (IPI on OpenVLA), 2025-12: 98.0 / 98.5 / 98.0 / 97.5",
       "level": "verified",
       "sources": [
        "s36"
       ],
       "note": "RL fine-tuning inside SimplerEnv; best checkpoint chosen by the same evaluation; 300 episodes.",
       "data": {
        "model": "STARE-VLA (IPI)",
        "date": "2025-12",
        "avg": 98,
        "rl": true
       }
      },
      {
       "value": "Google Robot best",
       "display": "Highest Google Robot numbers we found: Visual Matching 85.5 (Xiaomi-Robotics-0, 4-task average) and Variant Aggregation 84.9 (CORAL, 3-task average).",
       "level": "verified",
       "sources": [
        "s32",
        "s34"
       ],
       "note": "Not comparable across papers because averages cover 3 or 4 task groups."
      }
     ],
     "short": "97.9% average on WidowX (March 2026)"
    },
    "license_code": {
     "value": [
      "MIT",
      "Apache-2.0"
     ],
     "display": "MIT (SimplerEnv); Apache-2.0 (ManiSkill2_real2sim submodule with the environments and assets)",
     "level": "verified",
     "sources": [
      "s8",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE files read 2026-10-10."
    },
    "license_data": {
     "value": "MIT",
     "display": "MIT for the example evaluation videos on Hugging Face. No training data ships with SimplerEnv; the real-world reference results sit in the MIT-licensed code (metrics.py).",
     "level": "verified",
     "sources": [
      "s43",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "The RT-1 and BridgeData V2 training datasets carry their own licences (not checked here)."
    },
    "license_assets": {
     "value": "unknown",
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No asset licence list in the paper, README, guide or either repository tree (only the two root LICENSE files). The paper says objects come from Objaverse, from 3D scans of products bought on Amazon, from single-view 3D generation and from manual modelling; Variant Aggregation scenes are modified ReplicaCAD scenes. Objaverse objects each carry their own Creative Commons licence, some non-commercial; ReplicaCAD is CC BY 4.0 (sources s44, s45). Which Objaverse objects were used is not stated."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s7",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Install from GitHub; assets are in the repo; RT-1 checkpoints come from a public Google Cloud bucket and Octo from Hugging Face. Needs an NVIDIA GPU for SAPIEN rendering; no TPU support.",
     "short": "Open. The code is on GitHub."
    },
    "commercial_use": {
     "value": "unclear",
     "level": "inferred",
     "sources": [
      "s8",
      "s9",
      "s45",
      "s44",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Code is MIT and Apache-2.0, which allow commercial use. Some bundled objects come from Objaverse, where individual objects carry their own Creative Commons licences; the Objaverse card lists 25K CC BY-NC and 52K CC BY-NC-SA objects among them. The repo does not say which objects were used or under which terms. Not legal advice."
    },
    "sim_to_real": {
     "value": "correlated",
     "display": "Measured by its authors: Pearson r 0.924 (Google Robot) and 0.890 (WidowX) for 2023-2024 policies. Later checks with newer policies were weaker: r 0.548 by the AutoEval team on mostly new tasks, and r 0.402 by an independent group on SIMPLER-style versions of other tasks.",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s18",
      "s20",
      "s21",
      "s22",
      "s23",
      "s24",
      "s51"
     ],
     "checked": "2026-10-10",
     "note": "Level 'correlated' is our mapping; each study is listed under validity. Not marked 'replicated': the one fully independent study (Wang et al. 2026) built its own SIMPLER-style tasks for DROID hardware instead of using the official environments, and found weak agreement. The AutoEval and ManiSkill3 checks share authors with SIMPLER. Claims without paired numbers: PolaRiS (2025-12) says SIMPLER fails to give strong correlation for recent generalist policies, citing the OpenVLA paper, which does not mention SIMPLER in any of its three arXiv versions; PolaRiS ran no SIMPLER comparison and notes SIMPLER cannot handle wrist cameras. RobotArena Infinity (2025-10) found all tested VLAs scored much higher on its reproduction of the 4 SIMPLER WidowX scenes than on its 70 BridgeSim scenes, and suggests SIMPLER may overestimate performance; that compares two simulators. A GitHub user reported about 65% real versus about 10% simulated success for OpenVLA on put carrot on plate (issue #78; anecdotal).",
     "short": "Measured by its authors. Later checks found weaker agreement."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Scores come from simulation."
    },
    "citations": {
     "value": 578,
     "display": "578 (Semantic Scholar; 113 influential)",
     "level": "verified",
     "sources": [
      "s13"
     ],
     "checked": "2026-10-10",
     "note": "Read 2026-10-10 after several rate-limited attempts.",
     "short": "578"
    },
    "github_stars": {
     "value": 1174,
     "display": "1,174 stars, 201 forks (simpler-env/SimplerEnv)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "The community fork SimplerEnv-OpenVLA, which adds OpenVLA support, has 271 stars and 45 forks (s47).",
     "short": "1,174"
    },
    "used_by": {
     "value": "At least 207 papers reported SimplerEnv results by 2026-05-21, by our count of the 2026 audit's tracker. 24 of them have a first arXiv date in March 2026.",
     "level": "inferred",
     "sources": [
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Our count of rows labelled reports-results among 372 candidate rows in the released tracker CSV; the audit calls such counts lower bounds. By column: WidowX/Bridge numbers in 122 papers, Google Robot Visual Matching in 60, Variant Aggregation in 56; 67 rows report SimplerEnv results without naming the protocol.",
     "items": [
      {
       "value": "CogACT",
       "display": "2024-11. Google Robot and WidowX.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "SpatialVLA",
       "display": "2025-01. Google Robot and WidowX.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "Magma",
       "display": "Microsoft Research, 2025-02.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      },
      {
       "value": "ThinkAct",
       "display": "NVIDIA, 2025-07.",
       "level": "verified",
       "sources": [
        "s38"
       ]
      },
      {
       "value": "EO-1",
       "display": "2025-08.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "InternVLA-M1",
       "display": "Shanghai AI Laboratory, 2025-10.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "X-VLA",
       "display": "2025-10.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      },
      {
       "value": "Dexbotic",
       "display": "Dexmal and StepFun, 2025-10.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "Xiaomi-Robotics-0",
       "display": "Xiaomi, 2026-02.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "SimVLA and CORAL",
       "display": "Frontier Robotics, 2026-02 and 2026-03.",
       "level": "verified",
       "sources": [
        "s33",
        "s34"
       ]
      },
      {
       "value": "piRL and STARE-VLA",
       "display": "RL fine-tuning inside SimplerEnv, 2025.",
       "level": "verified",
       "sources": [
        "s35",
        "s36"
       ]
      },
      {
       "value": "CoVer-VLA",
       "display": "Stanford and NVIDIA, 2026-02. Uses custom SIMPLER tasks next to PolaRiS and real robots.",
       "level": "verified",
       "sources": [
        "s48"
       ]
      }
     ],
     "short": "At least 207 papers (May 2026)"
    },
    "industry_use": {
     "value": [
      "Microsoft",
      "NVIDIA",
      "Xiaomi",
      "Dexmal"
     ],
     "level": "verified",
     "sources": [
      "s37",
      "s38",
      "s32",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "Google DeepMind co-built SimplerEnv and ran the real Google Robot evaluations (facts.publishers).",
     "items": [
      {
       "value": "Microsoft",
       "display": "Magma (Microsoft Research) reports SimplerEnv results.",
       "level": "verified",
       "sources": [
        "s37"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "ThinkAct (NVIDIA) reports Google Robot and Bridge results.",
       "level": "verified",
       "sources": [
        "s38"
       ]
      },
      {
       "value": "Xiaomi",
       "display": "Xiaomi-Robotics-0 reports all three SimplerEnv settings.",
       "level": "verified",
       "sources": [
        "s32"
       ]
      },
      {
       "value": "Dexmal",
       "display": "Dexbotic toolbox (Dexmal, StepFun) reports SimplerEnv-Bridge results.",
       "level": "verified",
       "sources": [
        "s30"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "ManiSkill3 BridgeData v2 digital twins",
      "SimplerEnv-OpenVLA",
      "AutoEval SIMPLER scenes",
      "Audit stack-task variants"
     ],
     "display": "Ports, forks and variants built on SimplerEnv",
     "level": "verified",
     "sources": [
      "s19",
      "s47",
      "s20",
      "s14"
     ],
     "checked": "2026-10-10",
     "note": "Not a complete list.",
     "items": [
      {
       "value": "ManiSkill3 BridgeData v2 digital twins",
       "display": "GPU-parallel versions of the 4 WidowX tasks (2024-10).",
       "level": "verified",
       "sources": [
        "s19",
        "s18"
       ]
      },
      {
       "value": "SimplerEnv-OpenVLA",
       "display": "Community fork adding OpenVLA and other policies; 271 stars.",
       "level": "verified",
       "sources": [
        "s47"
       ]
      },
      {
       "value": "AutoEval SIMPLER scenes",
       "display": "A new drawer scene and a reverse eggplant-to-sink task that mirror the AutoEval cells (2025-03).",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Audit stack-task variants",
       "display": "Seven stack-task conditions released with the 2026 audit (2,016 episodes per policy).",
       "level": "verified",
       "sources": [
        "s14"
       ]
      }
     ],
     "short": "4 ports and variants built on it"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "other",
     "title": "Most claimed improvements are not shown to be statistically significant",
     "text": "The 2026 audit checked 122 previous-best-to-new comparisons on the WidowX protocol, using only published scores. Its paper classes 19.7% (24) as provably significant at the 5% level, 50.8% (62) as inconclusive, and the rest as no improvement or provably not significant. On 2026-10-08 the audit authors released corrected classifications: 27 of 122 provably significant (22.1% by our arithmetic), 62 inconclusive, 16 provably not significant and 17 no improvement.",
     "level": "verified",
     "sources": [
      "s14",
      "s15"
     ],
     "status": "open",
     "note": "The correction commit is titled 'Fix saturated-score significance classifications'. The paper (v1) has not been updated as of 2026-10-10.",
     "short": "Only about 1 in 5 claimed improvements on WidowX can be shown to be statistically significant from the published scores."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same model gets very different scores in different papers",
     "text": "WidowX averages reported for pi0 by other papers: 69.2% (EO-1, Xiaomi-Robotics-0), 66.7% (FASTer), 40.1% (SimVLA), 27.8% (X-VLA) and 27.1% (InternVLA-M1). For OpenVLA: 1.0% (SpatialVLA, EO-1), 4.2% (CogACT, InternVLA-M1), 7.8% (SimVLA), 8.3% (X-VLA) and 29.5% (FASTer). Google Robot averages cover 3 task groups in some papers and 4 in others: SpatialVLA reports its own Visual Matching score as 75.1% over 3 tasks, while EO-1 lists SpatialVLA at 55.3% over 4. Some tables do not add up: InternVLA-M1 lists pi0-FAST per-task scores that average 32.1% next to an average of 48.3%; SimVLA's WidowX table gives Octo-Base per-task scores of 12.5 / 8.3 / 0.0 / 43.1 with an average of 31.3. Xiaomi-Robotics-0 (2026-02) calls its 79.2% WidowX average the best result, while its comparison table leaves out X-VLA's 95.8% from 2025-10.",
     "level": "verified",
     "sources": [
      "s27",
      "s32",
      "s31",
      "s33",
      "s29",
      "s28",
      "s26",
      "s25"
     ],
     "status": "open",
     "short": "Different papers report pi0's WidowX average as anywhere from 27.1% to 69.2%."
    },
    {
     "id": "i3",
     "type": "other",
     "title": "Training data placed next to the test can match top scores",
     "text": "SimplerEnv tests in simulation, while its standard training data are real BridgeData V2 demonstrations, so a high score reads as transfer across that gap. The benchmark does not restrict training data. The 2026 audit trained a separate 22M-parameter policy per task, with no robotics pretraining, on 120 scripted demonstrations recorded in simulation next to the official test grid. Together the four policies scored 91/96 (94.8%), against 92/96 (95.8%) reported by X-VLA. The audit calls this an existence result: the score alone cannot tell crossing the gap from removing it. The same audit found no shortcut of the LIBERO kind: a small probe trained on the standard Bridge data scored 0.0%.",
     "level": "verified",
     "sources": [
      "s14"
     ],
     "status": "open",
     "short": "A 2026 audit trained small models (22 million parameters) on 120 scripted demonstrations per task, recorded in simulation next to the test positions. They scored 91 of 96 trials, against 92 of 96 reported by X-VLA."
    },
    {
     "id": "i4",
     "type": "protocol-variance",
     "title": "Independent reruns score lower than the published numbers",
     "text": "The 2026 audit reran five published WidowX policies on the official grid with the standard step limits (each of the 24 start states repeated 12 times; 1,152 episodes per policy). Results: CogACT-Base 48.7%, SpatialVLA 37.5%, InternVLA-M1 61.2%, X-VLA 72.4% and Dexbotic DB-MemVLA 64.7%. The papers report 51.3%, 42.7% (34.4% without fine-tuning), 71.7%, 95.8% and 84.4%. On the stack task alone, X-VLA reports 95.8% and the audit measured 59.7% (172/288). The audit says several of these policies report under longer, easier episode step limits.",
     "level": "verified",
     "sources": [
      "s17",
      "s14",
      "s25",
      "s26",
      "s28",
      "s29",
      "s30"
     ],
     "status": "open",
     "note": "The gaps are our arithmetic from the audit's released summaries and the papers. The audit does not say which SpatialVLA checkpoint it used.",
     "short": "When a 2026 audit reran five published policies, they scored 2.6 to 23.4 points below the published numbers."
    },
    {
     "id": "i5",
     "type": "protocol-variance",
     "title": "Results vary between runs and machines",
     "text": "The lead author advises at least 75 trials per Bridge task because the 24-trial grid is too noisy. A user saw 16% to 40% success for the same OpenVLA model on one task across runs. Another user's repeated RT-1-X runs differed from the paper and from each other, although RT-1-X outputs are deterministic; the lead author attributes this to simulator nondeterminism. The 2026 audit found SimplerEnv rollouts with CogACT diverged from step 0 when only the CPU or only the GPU was changed. A 2026 issue reports the task and object positions changing between runs with a fixed seed (no reply by 2026-10-10). The GPU (ManiSkill3) version is a separate implementation; its README says reproducing the paper needs the main branch.",
     "level": "verified",
     "sources": [
      "s39",
      "s40",
      "s14",
      "s41",
      "s11"
     ],
     "status": "open",
     "short": "One user saw the same model score between 16% and 40% on one task across runs."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "Small changes seen in the training data lower scores on the stack task",
     "text": "The 2026 audit changed the WidowX stack task in three ways that also occur in the BridgeData V2 training data: reversed colour order in the instruction, cubes starting on support blocks, and random cube and arm start poses (288 episodes per policy each). Stacked supports cut X-VLA from 172/288 to 90/288 (28.47 points, 95% CI 20.46 to 35.96) and Dexbotic from 129/288 to 79/288 (17.36 points). The reversed instruction cut CogACT by 11.11 points and InternVLA-M1 by 9.38. The audit reads these drops as overfitting to the fixed test grid.",
     "level": "verified",
     "sources": [
      "s14"
     ],
     "status": "open",
     "short": "When the cubes started on support blocks, X-VLA's success on the stack task fell from 59.7% to 31.3%."
    },
    {
     "id": "i7",
     "type": "saturated",
     "title": "Top WidowX scores are close to 100%",
     "text": "Since October 2025, X-VLA (95.8%), SimVLA (95.8%) and CORAL (97.9%) report WidowX averages above 95% without RL in the simulator, and STARE-VLA reports 98.0% after RL inside SimplerEnv. Several tasks have reported scores of 100%. Google Robot results are lower: the highest we found are 85.5% (Visual Matching) and 84.9% (Variant Aggregation).",
     "level": "verified",
     "sources": [
      "s29",
      "s33",
      "s34",
      "s36",
      "s32"
     ],
     "status": "open",
     "short": "By early 2026, top WidowX averages reached 95.8% to 97.9%."
    },
    {
     "id": "i8",
     "type": "other",
     "title": "Some top scores come from training inside the test simulator",
     "text": "piRL raised pi0 from 67.2% to 86.7% on the WidowX tasks with RL inside the simulator. STARE-VLA reports 98.0% after RL fine-tuning in SimplerEnv and reports the best checkpoint under the same evaluation it publishes. World-Gymnast trained RL baselines in SIMPLER copies of the AutoEval scenes and found they transferred poorly to the real cells: for example 34% versus 58% for its world-model method on opening the drawer.",
     "level": "verified",
     "sources": [
      "s35",
      "s36",
      "s49"
     ],
     "status": "open",
     "short": "Reinforcement learning (RL) inside the simulator raised pi0 from 67.2% to 86.7%."
    },
    {
     "id": "i9",
     "type": "inconsistent-reporting",
     "title": "The paper's own tables and versions disagree",
     "text": "In arXiv v1, Table I gives the Google Robot drawer task r 0.942 and MMRV 0.027 under Visual Matching, while Table IV and Fig. 6 give 0.915 and 0.055; the headline mean (r 0.924, MMRV 0.056) uses Table I. Fig. 8 gives MMRV 0.016 for the augmented RT-1 policy where Table VI gives 0.041. The camera-ready adds OpenVLA-7B to the Google Robot study and changes the per-task numbers (drawer r 0.823), but keeps the 6-checkpoint means in its Table 1. For put-eggplant-in-basket, Appendix B states 24 real trials, the real values imply 30, and the repo's metrics.py lists 0.250 and 0.400 for Octo-Base and Octo-Small where the paper has 0.233 and 0.433.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s12"
     ],
     "status": "open",
     "note": "The trial-count reading for the eggplant task is our arithmetic.",
     "short": "Different tables and versions of the SimplerEnv paper give different correlation numbers."
    },
    {
     "id": "i10",
     "type": "other",
     "title": "The real-robot check covered older policies on two setups",
     "text": "The authors' validation covers RT-1 checkpoints, RT-1-X, RT-2-X, Octo and, in the camera-ready, OpenVLA on the Google Robot, and RT-1-X and Octo on WidowX. Each per-task correlation rests on 3 to 7 points. The policies most reported today (pi0, pi0.5, GR00T, X-VLA) were not part of it. The real Google Robot evaluations were run by Google staff, so outside groups cannot repeat that comparison (our inference); the WidowX setup can be rebuilt. Visual Matching pastes real images behind a fixed camera view, so it cannot serve policies that use wrist cameras, which most current generalist policies do (PolaRiS paper).",
     "level": "inferred",
     "sources": [
      "s2",
      "s4",
      "s50",
      "s22"
     ],
     "status": "open",
     "short": "The authors' real-robot check covered policies from the RT-1 era on two setups. The policies most reported today were not part of it."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A SimplerEnv score is good evidence of real-robot ranking only for policies like those its authors tested (RT-1, RT-2-X, Octo, OpenVLA) on these two setups. For current VLA models the evidence is weak. The two later studies that included newer policies found r = 0.548 and r = 0.402, both mostly on tasks outside the official set.",
     "basis": [
      "facts.sim_to_real",
      "issues.i10"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Scores agreed with real-robot results for older policies. The evidence for current VLA models is weak."
    },
    {
     "id": "r2",
     "text": "Do not compare WidowX numbers across papers without checking trials, step limits, training data and which baseline values were copied. Reported numbers for the same model differ by up to 42 points, and independent reruns came in lower than published.",
     "basis": [
      "issues.i2",
      "issues.i4",
      "issues.i5",
      "facts.trials"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the protocol before comparing numbers across papers."
    },
    {
     "id": "r3",
     "text": "Above about 95% on WidowX, scores no longer separate methods, and such scores can be reached by training close to the test or inside the simulator. Scores on the Google Robot settings are further from 100%.",
     "basis": [
      "issues.i7",
      "issues.i3",
      "issues.i8",
      "issues.i1"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Top WidowX scores say little about real-world generalisation."
    },
    {
     "id": "r4",
     "text": "SimplerEnv's method for checking a simulator against real robots has become the common standard. It scores the same policies in simulation and on real robots and compares the results with Pearson r (a correlation) and MMRV (a measure of how often two rankings disagree). AutoEval, PolaRiS and later studies report the same two numbers.",
     "basis": [
      "facts.sim_to_real",
      "sources.s20",
      "sources.s22",
      "sources.s21"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Its method for checking a simulator against real robots has become the common standard."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well current vision-language-action (VLA) models will do on real robots.",
     "sub": "The authors validated it on policies from 2023 and 2024. Later checks with newer policies found weaker agreement.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether a small improvement is real.",
     "sub": "Most claimed improvements cannot be shown to be statistically significant.",
     "basis": [
      "issues.i1",
      "issues.i5"
     ]
    },
    {
     "id": "l3",
     "text": "How well a policy generalises beyond its training data.",
     "sub": "Policies trained on data close to the test, or inside the test simulator, can reach top scores.",
     "basis": [
      "issues.i3",
      "issues.i8"
     ]
    }
   ],
   "validity": [
    {
     "id": "v1",
     "name": "SIMPLER paper: Google Robot, Visual Matching",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 6 checkpoints (3 RT-1 checkpoints, RT-1-X, RT-2-X and Octo-Base, so 4 distinct models) were scored in simulation and on the real Google Robot, on the same tasks and trial grids. Agreement was measured with Pearson r and MMRV (a measure of how often two rankings disagree).",
     "result": "Pearson r = 0.924 and MMRV = 0.056 (mean of 3 task groups)",
     "authors_view": "strong",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Table I. Per task (Fig. 6): pick coke can r 0.976 / MMRV 0.031, move near 0.855 / 0.111, open/close drawer 0.915 / 0.055, drawer + apple 0.969 / 0.000. Table I lists the drawer as 0.942 / 0.027 (issues.i9); with Table IV's drawer values the 3-task mean would be r 0.915, MMRV 0.066 (our arithmetic)."
    },
    {
     "id": "v2",
     "name": "SIMPLER paper: Google Robot, Variant Aggregation",
     "date": "2024-05",
     "by": "authors",
     "method": "The same 6 checkpoints were scored in simulation with randomised backgrounds, lighting, distractors and table textures, and compared with the same real results.",
     "result": "Pearson r = 0.778 and MMRV = 0.143 (mean of 3 task groups)",
     "n_policies": 6,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Table I. The authors recommend Visual Matching as the default because Variant Aggregation correlated less well."
    },
    {
     "id": "v3",
     "name": "SIMPLER paper: WidowX and BridgeData V2",
     "date": "2024-05",
     "by": "authors",
     "method": "3 policies (RT-1-X, Octo-Base and Octo-Small, so 2 model families) were scored in simulation and on a real WidowX on the same 4 tasks, with 24 real trials per task.",
     "result": "Pearson r = 0.890 and MMRV = 0.014 (mean over 4 tasks, for success and grasp)",
     "authors_view": "strong",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Fig. 7 averages 8 columns, the grasp and success rates for each task (our arithmetic from Table V). Success only: spoon r 0.827, carrot 0.575, stack 1.000, eggplant 0.990; MMRV 0 except carrot 0.111. With 3 policies each r rests on 3 points."
    },
    {
     "id": "v4",
     "name": "SIMPLER camera-ready: Google Robot with OpenVLA added",
     "date": "2025-01",
     "by": "authors",
     "method": "7 checkpoints (the 6 above plus OpenVLA-7B, so 5 distinct models) were scored in simulation (Visual Matching) and on the real Google Robot, on the same tasks.",
     "result": "Pearson r = 0.929 and MMRV = 0.049 (Fig. 4)",
     "authors_view": "strong",
     "n_policies": 7,
     "level": "verified",
     "sources": [
      "s4"
     ],
     "note": "Per task (Table 2): pick coke can r 0.969 / MMRV 0.027, move near 0.864 / 0.095, open/close drawer 0.823 / 0.177, drawer + apple 0.973 / 0.000. Adding OpenVLA lowered the drawer correlation from 0.915 to 0.823. How the overall Fig. 4 number is pooled is not stated."
    },
    {
     "id": "v5",
     "name": "SIMPLER paper: Isaac Sim re-implementation",
     "date": "2024-05",
     "by": "authors",
     "method": "5 checkpoints (3 RT-1 checkpoints, RT-1-X and Octo-Base) were scored on the pick coke can and move near tasks in Variant Aggregation scenes rebuilt in Isaac Sim, and compared with the same real results.",
     "result": "Pearson r = 0.919 and MMRV = 0.058. In SAPIEN, r = 0.923 and MMRV = 0.082.",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Fig. 10 and Table XI. Tests whether the method carries over to another simulator; the released code is the SAPIEN version."
    },
    {
     "id": "v6",
     "name": "SIMPLER paper: sensitivity to visual shifts",
     "date": "2024-05",
     "by": "authors",
     "method": "2 RT-1 policies, with and without image augmentation, were tested under 5 visual shifts. Their drop in success in simulation was compared with earlier real-world tests by Xie et al.",
     "result": "Pearson r = 0.831 and MMRV = 0.000 without augmentation. Pearson r = 0.970 and MMRV = 0.016 with augmentation.",
     "authors_view": "accurately reflect",
     "n_policies": 2,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "note": "Fig. 8. Compares sensitivity patterns within each policy, not rankings across policies. Table VI gives MMRV 0.041 for the augmented policy."
    },
    {
     "id": "v7",
     "name": "ManiSkill3 GPU port",
     "date": "2024-10",
     "by": "authors",
     "method": "3 policies (Octo-Base, Octo-Small and RT-1-X) were run on the GPU version of the 4 WidowX tasks and compared with the SIMPLER paper's real results. Success and grasp rates were pooled into one plot.",
     "result": "Pearson r = 0.9284 and MMRV = 0.0147",
     "authors_view": "close to that of the original paper",
     "n_policies": 3,
     "level": "verified",
     "sources": [
      "s18"
     ],
     "note": "Shares authors with SIMPLER (Xuanlin Li, Hao Su). Reuses the paper's real-world numbers, so it checks the port, not new real runs."
    },
    {
     "id": "v8",
     "name": "AutoEval paper (UC Berkeley)",
     "date": "2025-03",
     "by": "authors",
     "method": "6 policies (OpenVLA, Octo, Open-pi0, MiniVLA, SuSIE and SuSIE's low-level policy, so 5 distinct models) were run 50 times per task in SIMPLER and in human-run real evaluations on matching WidowX cells. Of the 4 tasks, only eggplant-in-basket is an official SimplerEnv task.",
     "result": "Mean Pearson r = 0.548 and MMRV = 0.207 over 4 tasks (our computation from the paper's tables)",
     "authors_view": "policy dependent",
     "n_policies": 6,
     "level": "inferred",
     "sources": [
      "s20"
     ],
     "note": "Not independent: 2 of the 5 AutoEval authors (Karl Pertsch, Sergey Levine) are SIMPLER co-authors. The other 3 tasks are a reverse eggplant-to-sink task in the same scene and 2 drawer tasks in a scene the AutoEval team built with SIMPLER's guide. The paper plots these values as bars without printing them (Fig. 7); our method reproduces its printed AutoEval numbers exactly (r 0.942, MMRV 0.015). Per task: open drawer r 0.942 / MMRV 0.153, close drawer 0.267 / 0.457, eggplant to basket 0.131 / 0.217, eggplant to sink 0.850 / 0.000. Example: Open-pi0 put eggplant in sink 6/50 in SIMPLER, 47/50 on the real robot."
    },
    {
     "id": "v9",
     "name": "Wang et al. (Tsinghua, Shanghai Qi Zhi)",
     "date": "2026-06",
     "by": "independent",
     "method": "5 VLA policies (pi0, pi0-FAST, pi0.5, GR00T N1.6 and GR00T N1.7) were run on 7 tabletop tasks rebuilt in SIMPLER's SAPIEN setting and matched to real tasks on DROID hardware, under changes to vision, layout and language. Each policy had 20 simulated and 5 real test runs per task and change.",
     "result": "Mean Spearman correlation 0.400, Pearson r = 0.402 and MMRV = 0.128",
     "n_policies": 5,
     "level": "verified",
     "sources": [
      "s21"
     ],
     "note": "Not SimplerEnv's official Google Robot or WidowX tasks. Per change: vision rho 0.700 / r 0.701 / MMRV 0.102, layout 0.300 / 0.420 / 0.056, language 0.200 / 0.086 / 0.226. In the same study REALM scored 0.700 / 0.785 / 0.030 and VLA-Arena 0.575 / 0.725 / 0.060."
    }
   ],
   "searched": [
    {
     "for": "validity (paired sim-vs-real studies of SimplerEnv)",
     "where": "SIMPLER arXiv v1 and PMLR camera-ready (all tables); ManiSkill3 paper v1 and v2; AutoEval paper v2 and camera-ready; Wang et al. 2606.10366; PolaRiS v1, v2 and RSS 2026 version (no SIMPLER comparison run); OpenVLA v1-v3 (no mention of SIMPLER, although PolaRiS cites it for SIMPLER's weak correlation); RobotArena Infinity (sim-vs-sim only; its real-world check of one task did not involve SIMPLER); WorldEval 2505.19017 (applied SIMPLER's techniques to RoboTwin, said a direct SIMPLER comparison was not feasible); Scalable Policy Evaluation with Video World Models 2511.11520, WorldGym 2506.00613, Ctrl-World 2510.10125, Veo world simulator 2512.10675, GSWorld 2510.20813, Real-is-Sim 2504.03597, soft-body splat evaluation 2511.04665, RoboArena 2506.18123, SimFoundry 2606.28276, H2RBench 2609.24778 (none pairs SimplerEnv with real results); EchoArena (CVPR 2026 workshop; compares a world model with SimplerEnv, not with real robots); SimplerEnv GitHub issues; web searches (extended) for SimplerEnv sim-real correlation studies.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "Paper Appendix C, README, ADDING_NEW_ENVS_ROBOTS.md, full git trees of SimplerEnv and ManiSkill2_real2sim (only root LICENSE files), Hugging Face cards for ReplicaCAD and Objaverse.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project site, README, GitHub repo; the 2026 audit's tracker is a compilation, not an official board.",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "2026 audit tracker and previous-SOTA export (WidowX column) and the 14 papers listed in sources; values read in each paper. Not searched beyond the audit snapshot (2026-05-21) except papers already found.",
     "date": "2026-10-10"
    },
    {
     "for": "uncertainty_reported",
     "where": "SimplerEnv tables in CogACT, SpatialVLA, Magma, ThinkAct, EO-1, InternVLA-M1, X-VLA, Dexbotic, FASTer, Xiaomi-Robotics-0, SimVLA, CORAL, piRL and STARE-VLA.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Evaluating Real-World Robot Manipulation Policies in Simulation (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2405.05941",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "SIMPLER paper, full text arXiv v1 (Tables I-XIV, Appendix B)",
     "url": "https://arxiv.org/pdf/2405.05941v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "SIMPLER, PMLR proceedings page (CoRL 2024, PMLR 270:3705-3728)",
     "url": "https://proceedings.mlr.press/v270/li25c.html",
     "type": "paper",
     "publisher": "PMLR (8th Conference on Robot Learning)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "SIMPLER camera-ready PDF in PMLR (Fig. 4, Tables 1-3, adds OpenVLA-7B)",
     "url": "https://raw.githubusercontent.com/mlresearch/v270/main/assets/li25c/li25c.pdf",
     "type": "paper",
     "publisher": "PMLR (8th Conference on Robot Learning)",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "SIMPLER project site",
     "url": "https://simpler-env.github.io/",
     "type": "site",
     "publisher": "SIMPLER team (UCSD, Stanford, UC Berkeley, Google DeepMind)",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "GitHub API: simpler-env/SimplerEnv (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/simpler-env/SimplerEnv",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "SimplerEnv README (main branch)",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/main/README.md",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "SimplerEnv LICENSE (MIT)",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/main/LICENSE",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024-03",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "ManiSkill2_real2sim submodule: LICENSE (Apache-2.0), README, setup.py (0.5.3, sapien==2.2.2), asset folders",
     "url": "https://github.com/simpler-env/ManiSkill2_real2sim",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "SimplerEnv commit history, tags and releases (main branch)",
     "url": "https://github.com/simpler-env/SimplerEnv/commits/main",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2026-09-30",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "SimplerEnv maniskill3 branch README and commits",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/maniskill3/README.md",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "SimplerEnv metrics.py (REAL_PERF and SIMPLER_PERF tables)",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/main/simpler_env/utils/metrics.py",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "Semantic Scholar API record for arXiv:2405.05941",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2405.05941?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "What Are We Actually Benchmarking in Robot Manipulation? (2026 audit; Sections 4-6, Appendix A.2, A.3, B)",
     "url": "https://arxiv.org/abs/2606.04233",
     "type": "paper",
     "publisher": "arXiv (TTIC, University of Chicago, Argonne); CoRL 2026 per project page",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Audit repository: statistical-significance counts, paper vs corrected release (commit fe18d73, 2026-10-08)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/analysis/release_current_values/statistical_significance_bucket_comparison.csv",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026-10-08",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Audit repository: SimplerEnv citation tracker CSV (snapshot used for the paper, 2026-05-21)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/blob/main/leaderboards/simplerenv/simplerenv_citation_tracker.csv",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Audit repository: SimplerEnv fixed-grid calibration reruns (per_policy_summary.csv, per_task_summary.csv)",
     "url": "https://github.com/ripl/ManipulationBenchmarkAudit/tree/main/creeping_overfitting/results/simplerenv/fixed_grid_calibration",
     "type": "repo",
     "publisher": "TTIC RIPL",
     "date": "2026-09",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI (Section III-E, Appendix K, Fig. 25)",
     "url": "https://arxiv.org/abs/2410.00425",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-10",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "ManiSkill documentation: Digital Twins (BridgeData v2 evaluation environments)",
     "url": "https://github.com/haosulab/ManiSkill/blob/main/docs/source/tasks/digital_twins/index.md",
     "type": "repo",
     "publisher": "Hao Su Lab (UCSD)",
     "date": "2025",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World (v2; Section 5.2, Fig. 7, Appendix C-D, Tables 1-3)",
     "url": "https://arxiv.org/html/2503.24278v2",
     "type": "paper",
     "publisher": "arXiv (UC Berkeley, NVIDIA)",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation (Table 2, Appendix A)",
     "url": "https://arxiv.org/abs/2606.10366",
     "type": "paper",
     "publisher": "arXiv (Tsinghua University, Shanghai Qi Zhi Institute)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (v2; Sections 2 and 5.1)",
     "url": "https://arxiv.org/html/2512.16881v2",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "OpenVLA: An Open-Source Vision-Language-Action Model (v1-v3 full text checked for SIMPLER)",
     "url": "https://arxiv.org/abs/2406.09246",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-06",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "RobotArena Infinity: Scalable Robot Benchmarking via Real-to-Sim Translation (Sections 5.2-5.3)",
     "url": "https://arxiv.org/abs/2510.23571",
     "type": "paper",
     "publisher": "arXiv; ICLR 2026",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "CogACT (Tables 1-2: SIMPLER Google Robot and WidowX)",
     "url": "https://arxiv.org/abs/2411.19650",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-11",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "SpatialVLA (Tables I-II: SimplerEnv)",
     "url": "https://arxiv.org/abs/2501.15830",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "EO-1: An Open Unified Embodied Foundation Model for General Robot Control (Table 4: SimplerEnv)",
     "url": "https://arxiv.org/abs/2508.21112",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "InternVLA-M1 (Tables 1-2: SimplerEnv)",
     "url": "https://arxiv.org/abs/2510.13778",
     "type": "paper",
     "publisher": "arXiv (Intern Robotics, Shanghai AI Laboratory)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "X-VLA (Table 2 and Table 12: Simpler)",
     "url": "https://arxiv.org/abs/2510.10274",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Dexbotic: Open-Source Vision-Language-Action Toolbox (Table 1: SimplerEnv-Bridge)",
     "url": "https://arxiv.org/abs/2510.23511",
     "type": "paper",
     "publisher": "arXiv (Dexmal, StepFun)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "FASTer: Toward Efficient Autoregressive Vision Language Action Modeling (Table 1 and Appendix: Simpler-Bridge)",
     "url": "https://arxiv.org/abs/2512.04952",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Xiaomi-Robotics-0 (Section 4 and Appendix B: SimplerEnv)",
     "url": "https://arxiv.org/abs/2602.12684",
     "type": "paper",
     "publisher": "Xiaomi Robotics",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "SimVLA: A Simple VLA Baseline for Robotic Manipulation (Tables 4-5)",
     "url": "https://arxiv.org/abs/2602.18224",
     "type": "paper",
     "publisher": "arXiv (Frontier Robotics)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "CORAL: Scalable Multi-Task Robot Learning via LoRA Experts (Tables II-III)",
     "url": "https://arxiv.org/abs/2603.09298",
     "type": "paper",
     "publisher": "arXiv (Frontier Robotics)",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "piRL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models (Table 8, Appendix E.1)",
     "url": "https://arxiv.org/abs/2510.25889",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "STARE-VLA: Progressive Stage-Aware Reinforcement for Fine-Tuning VLA Models (Table 1, Appendix A.4-A.5)",
     "url": "https://arxiv.org/abs/2512.05107",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "Magma: A Foundation Model for Multimodal AI Agents (SimplerEnv results)",
     "url": "https://arxiv.org/abs/2502.13130",
     "type": "paper",
     "publisher": "arXiv (Microsoft Research et al.)",
     "date": "2025-02",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "ThinkAct: Vision-Language-Action Reasoning via Reinforced Visual Latent Planning (Table 1: SimplerEnv)",
     "url": "https://arxiv.org/abs/2507.16815",
     "type": "paper",
     "publisher": "arXiv (NVIDIA, National Taiwan University)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "SimplerEnv issue #108: Unstable success rates for evaluating OpenVLA (reply by lead author)",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/108",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "SimplerEnv issue #125: Different results of configuration evaluation (reply by lead author)",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/125",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "SimplerEnv issue #130: Inconsistent Simulation Environment (no reply)",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/130",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s42": {
     "title": "SimplerEnv issue #121: Moving ManiSkill2_real2sim tasks to ManiSkill3 (reply by lead author)",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/121",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s43": {
     "title": "Hugging Face dataset card: xuanlinli17/simpler-env-eval-example-videos (license: mit)",
     "url": "https://huggingface.co/datasets/xuanlinli17/simpler-env-eval-example-videos",
     "type": "repo",
     "publisher": "Xuanlin Li (lead author)",
     "date": "2024-04",
     "accessed": "2026-10-10"
    },
    "s44": {
     "title": "ReplicaCAD dataset card (CC BY 4.0)",
     "url": "https://huggingface.co/datasets/ai-habitat/ReplicaCAD_dataset",
     "type": "repo",
     "publisher": "AI Habitat (Meta)",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s45": {
     "title": "Objaverse dataset card (ODC-By; per-object Creative Commons licences)",
     "url": "https://huggingface.co/datasets/allenai/objaverse",
     "type": "repo",
     "publisher": "Allen Institute for AI",
     "date": "2023",
     "accessed": "2026-10-10"
    },
    "s46": {
     "title": "SimplerEnv guide: ADDING_NEW_ENVS_ROBOTS.md (ReplicaCAD scenes modified in Blender)",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/main/ADDING_NEW_ENVS_ROBOTS.md",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024-05",
     "accessed": "2026-10-10"
    },
    "s47": {
     "title": "SimplerEnv-OpenVLA (community fork adding OpenVLA support; GitHub API)",
     "url": "https://github.com/DelinQu/SimplerEnv-OpenVLA",
     "type": "repo",
     "publisher": "Delin Qu",
     "date": "2025-06",
     "accessed": "2026-10-10"
    },
    "s48": {
     "title": "Scaling Verification Can Be More Effective than Scaling Policy Learning for VLA Alignment (CoVer-VLA)",
     "url": "https://arxiv.org/abs/2602.12281",
     "type": "paper",
     "publisher": "arXiv (Stanford, NVIDIA Research)",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s49": {
     "title": "World-Gymnast: Training Robots with Reinforcement Learning in a World Model (Table 1, Appendix B.2)",
     "url": "https://arxiv.org/abs/2602.02454",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-02",
     "accessed": "2026-10-10"
    },
    "s50": {
     "title": "SimplerEnv issue #77: Source of Real Evaluation Results (reply by lead author)",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/77",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s51": {
     "title": "SimplerEnv issue #78: Poor performance of OpenVLA on Bridge",
     "url": "https://github.com/simpler-env/SimplerEnv/issues/78",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2025-03",
     "accessed": "2026-10-10"
    },
    "s52": {
     "title": "SimplerEnv setup.py (simpler_env 0.0.1)",
     "url": "https://github.com/simpler-env/SimplerEnv/blob/main/setup.py",
     "type": "repo",
     "publisher": "simpler-env",
     "date": "2024",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created at full depth from primary sources, starting from the basic entry and the real-eval inventory record. Added the camera-ready validation numbers, the ManiSkill3, AutoEval and Wang et al. comparisons, the audit's corrected significance counts and reruns, verified top scores and adoption counts."
    },
    {
     "date": "2026-10-11",
     "change": "Published as a full entry."
    }
   ]
  },
  {
   "id": "teach",
   "name": "TEACh",
   "aliases": [
    "Task-driven Embodied Agents that Chat",
    "TEACh EDH",
    "TEACh TfD",
    "TEACh TATC"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Defines benchmarks (EDH, TfD, TATC) that score a simulated household agent acting in AI2-THOR from dialogue.",
   "summary": {
    "text": "Dataset and benchmarks of human-human dialogues for completing household tasks together in the AI2-THOR simulator.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Amazon Alexa AI; USC Viterbi Department of Computer Science; University of Michigan EECS",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Industry research lab (Amazon Alexa AI); see taxonomy_friction."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2021-10 (arXiv v1 2021-10-01)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "value": "README: 'As of 09/07/2022' dataset updated with dialog-act annotations; last commit 2023-11-01",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Commit date from GitHub API."
    },
    "version": {
     "value": "arXiv v3 (2021-12-29) uses a new EDH test set restricted to task-relevant state changes",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Conflict: the README says v3 'will be public on Dec 30, 2022'; arXiv shows v3 submitted 2021-12-29. arXiv date taken as correct."
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "instruction-following",
      "collaboration",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "virtual-agent"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "kitchen",
      "home"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scale": {
     "value": "3,047 successful sessions (4,365 collected, 3,320 human-successful = 74.17%; collection cost $105k); 12 task types, 438 unique variants; 45k+ utterances",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "3,047 vs 3,320: the paper says some successful sessions were dropped from the benchmarks due to replay issues."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "top_score": {
     "value": "Paper baselines (adapted E.T.): EDH val unseen SR 7.83% (+S); EDH test unseen SR 5.68% (+H)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "Papers only (. README describes a 'TEACh Benchmark Challenge' with Docker submissions but names no public leaderboard; none found on EvalAI.)",
     "note": "Searched full EvalAI challenge list (2026-10-10) and web; no TEACh leaderboard found."
    },
    "license_code": {
     "value": "MIT (SOFTWARELICENSE: code and model weights)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "Non-image data files: Community Data License Agreement - Sharing 1.0 (DATALICENSE); images: Apache 2.0 following AI2-THOR (IMAGESLICENSE)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Images: https://github.com/alexa/teach/blob/main/IMAGESLICENSE"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "No real-robot evaluation found in the paper or repository."
    },
    "citations": {
     "value": 305,
     "display": "305 (Semantic Scholar)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "github_stars": {
     "value": 145,
     "display": "145 (alexa/teach)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10"
    },
    "status": {
     "value": "dormant",
     "level": "inferred",
     "sources": [
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "No substantive change since 2022; last commit is a 2023 dependency bump."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "TEACh: Task-driven Embodied Agents that Chat",
     "url": "https://arxiv.org/abs/2110.00534",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-10"
    },
    "s2": {
     "title": "TEACh: Task-driven Embodied Agents that Chat (full text)",
     "url": "https://arxiv.org/html/2110.00534v3",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2021-10"
    },
    "s3": {
     "title": "alexa/teach on GitHub (file README.md)",
     "url": "https://github.com/alexa/teach/blob/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "alexa/teach on GitHub (blob)",
     "url": "https://github.com/alexa/teach/blob/main/SOFTWARELICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "alexa/teach on GitHub (blob)",
     "url": "https://github.com/alexa/teach/blob/main/DATALICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/ARXIV:2110.00534?fields=citationCount",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s7": {
     "title": "alexa/teach on GitHub (repository)",
     "url": "https://api.github.com/repos/alexa/teach",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "alexa/teach on GitHub (repository)",
     "url": "https://github.com/alexa/teach",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "umi-bench",
   "name": "UMI-Bench",
   "full_name": "UMI-Bench 1.0",
   "aliases": [
    "UMI-Bench",
    "UMI-Benchmark-v1"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Scores manipulation policies by closed-loop rollouts on real robots.",
   "summary": {
    "text": "Real-robot benchmark of 10 tabletop tasks for policies trained on handheld UMI-style gripper data.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Soochow University; Lumos Robotics; Fudan University; Shanghai Jiao Tong University; Shanghai TeleAI; Shanghai AI Laboratory; INSAIT; Xi'an Jiaotong-Liverpool University. 19 authors.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Affiliations from project page; arXiv abs page lists authors only. Checked 2026-10-10."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Mostly universities and labs, one company (Lumos Robotics). Checked 2026-10-10."
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "7 of 8 listed organisations are in China; INSAIT is in Bulgaria. Checked 2026-10-10."
    },
    "first_release": {
     "value": "2026-06 (arXiv v1 2026-06-09; Hugging Face dataset created 2026-06-03)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "latest_update": {
     "value": "Hugging Face dataset last modified 2026-09-28; no newer paper version",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Hugging Face API lastModified. Checked 2026-10-10."
    },
    "version": {
     "value": "1.0",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "published_at": {
     "value": "arXiv preprint (no comments or journal-ref)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "venue": {
     "value": "real",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "capability": {
     "value": [
      "manipulation",
      "bimanual",
      "long-horizon"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Tasks include stacking, articulated container, tool-mediated stamping, slot insertion, bimanual pouring, packing, dynamic pick on a turntable, sorting, long-horizon rearrangement, deformable folding. Checked 2026-10-10."
    },
    "embodiment": {
     "value": [
      "single-arm",
      "bimanual-arm"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "robots": {
     "value": "FastTouch tabletop arm (6-DoF URDF and Python SDK released); data collected with FastUMI Pro handheld device; wrist RGB cameras",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "scene": {
     "value": [
      "tabletop"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "1.2 m x 1.0 m x 0.75 m table with a 5 cm grid board. Checked 2026-10-10."
    },
    "scoring": {
     "value": [
      "success-rate",
      "progress",
      "human-rating"
     ],
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Full Success Rate (binary) and Progress Score (0-100, task rubrics); human evaluators annotate recorded rollouts. 'Overall Score' aggregates 50 rollouts per task-model pair; exact formula not given. Checked 2026-10-10."
    },
    "trials": {
     "value": "50 real-world rollouts per task (500 total per model)",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "evaluator": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "display": "self-reported: local-first protocol run by each lab; paper results run by the authors",
     "note": "Checked 2026-10-10."
    },
    "top_score": {
     "value": "Overall Score: pi0.5 55.84, pi0 48.90, DreamZero 40.59 (paper); T3 and T9 at 0% FSR for all three",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "leaderboard": {
     "value": "none",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "None (: paper says result packages suffice for 'later inspection and leaderboard submission' but no leaderboard URL exists; project page has none)",
     "note": "Checked 2026-10-10."
    },
    "license_code": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Also searched GitHub for UMI-Bench repos; found only unaffiliated personal repos. Checked 2026-10-10."
    },
    "license_data": {
     "value": "CC-BY-4.0 (Hugging Face dataset card)",
     "level": "verified",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Checked 2026-10-10."
    },
    "commercial_use": {
     "value": "allowed",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "display": "Allowed (data)",
     "note": "CC BY 4.0. Not legal advice. Checked 2026-10-10."
    },
    "sim_to_real": {
     "value": "not-applicable",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Scores on real robots. Real reproducibility: protocol-only (no multi-site study)."
    },
    "real_reproducibility": {
     "value": null,
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "protocol-only: protocol meant to let other sites rebuild the setup; no multi-site study reported. Paper lists operator reset variation, calibration drift, lighting and timing as known noise sources.",
     "note": "Checked 2026-10-10."
    },
    "citations": {
     "value": 2,
     "display": "2",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "Semantic Scholar. Checked 2026-10-10."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "UMI-Bench 1.0: An Open and Reproducible Real-World Benchmark for Tabletop Robotic Manipulation with UMI Data",
     "url": "https://arxiv.org/abs/2606.10382",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    },
    "s2": {
     "title": "UMI-Bench 1.0",
     "url": "https://umibenchmark.github.io/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "UMIbenchmark/UMI-Benchmark-v1 on Hugging Face (dataset)",
     "url": "https://huggingface.co/datasets/UMIbenchmark/UMI-Benchmark-v1",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s4": {
     "title": "UMI-Bench 1.0: An Open and Reproducible Real-World Benchmark for Tabletop Robotic Manipulation with UMI Data (full text)",
     "url": "https://arxiv.org/html/2606.10382v1",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2026-06"
    },
    "s5": {
     "title": "UMIbenchmark/UMI-Benchmark-v1-checkpoints on Hugging Face (model)",
     "url": "https://huggingface.co/UMIbenchmark/UMI-Benchmark-v1-checkpoints",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s6": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2606.10382",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "vlabench",
   "name": "VLABench",
   "aliases": [],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed simulated manipulation tasks and evaluation episodes produce progress/success scores for VLA policies; a second track scores VLMs on embodied task planning.",
   "summary": {
    "text": "Simulated single-arm benchmark of 100 language-driven manipulation task types testing common sense, semantics and long-horizon planning.",
    "sources": [
     "s6"
    ]
   },
   "facts": {
    "publishers": {
     "value": "All 11 authors list School of Computer Science, Fudan University; corresponding authors include Xipeng Qiu",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Code is hosted under the OpenMOSS GitHub organisation."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "china",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2024-12 (arXiv v1 2024-12-24; README: preview version released 2024/12/25)",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "latest_update": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "README 2025/11/10: new fine-tuned baseline checkpoints; last commit 2025-11-11",
       "level": "verified",
       "sources": [
        "s3"
       ]
      },
      {
       "value": "Hugging Face dataset VLABench/vlabench_primitive_pretrain_lerobot last modified 2026-07-07",
       "level": "verified",
       "sources": [
        "s4"
       ]
      }
     ]
    },
    "version": {
     "value": "Preview; no releases or tags",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Tags/releases checked via GitHub API: none."
    },
    "published_at": {
     "value": "ICCV 2025, pp. 11142-11152 (README: accepted 2025/6/26)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "capability": {
     "value": [
      "manipulation",
      "instruction-following",
      "long-horizon",
      "embodied-reasoning"
     ],
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "embodiment": {
     "value": [
      "single-arm"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "tabletop",
      "kitchen",
      "home",
      "retail-logistics",
      "office-lab"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Tasks are tabletop manipulation set inside these scenes."
    },
    "scoring": {
     "value": [
      "progress",
      "success-rate",
      "composite"
     ],
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "display": "Papers only (: project page shows full leaderboard 'Coming Soon'; README TODO 'Leaderboard of VLAs and VLMs' unchecked; README lists Track 1 SR for team baselines (π0 47%, π0.5 40.6%)"
    },
    "license_code": {
     "value": "MIT (LICENSE.txt); copyright line reads 'Copyright (c) 2024 simpler-env', apparently copied from another project",
     "level": "verified",
     "sources": [
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "license_data": {
     "value": "MIT tag on 7 of 10 VLABench Hugging Face datasets incl. VLABench/assets; Apache-2.0 on vlabench_composite_ft_lerobot_video; no tag on vlm_evaluation_v1.0 and vlabench_primitive_ft_lerobot",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "commercial_use": {
     "value": null,
     "level": "inferred",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "display": "allowed for code and MIT-tagged data; unclear for untagged datasets and third-party 3D assets",
     "note": "Paper says some assets were generated with AI tools; not legal advice."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "Not checked (: no real-robot experiments in the paper; no follow-up measuring VLABench vs real found)",
     "note": "Paper has no real-robot experiments. README 'Preview' note promises real-device deployment with a future release. A June 2026 sim-and-real correlation study (arXiv 2606.10366) cites VLABench but measures VLA-Arena, SIMPLER and REALM, not VLABench."
    },
    "status": {
     "value": "maintained",
     "level": "inferred",
     "sources": [
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "No code commits since 2025-11; datasets updated 2026-07."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "VLABench: A Large-Scale Benchmark for Language-Conditioned Robotics Manipulation with Long-Horizon Reasoning Tasks (full text)",
     "url": "https://arxiv.org/html/2412.18194v1",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-12"
    },
    "s2": {
     "title": "VLABench: A Large-Scale Benchmark for Language-Conditioned Robotics Manipulation with Long-Horizon Reasoning Tasks",
     "url": "https://arxiv.org/abs/2412.18194",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2024-12"
    },
    "s3": {
     "title": "OpenMOSS/VLABench on GitHub (commits?per_page=3)",
     "url": "https://api.github.com/repos/OpenMOSS/VLABench/commits?per_page=3",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s4": {
     "title": "api/datasets on Hugging Face (model)",
     "url": "https://huggingface.co/api/datasets?author=VLABench",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Hugging Face"
    },
    "s5": {
     "title": "VLABench",
     "url": "https://vlabench.github.io/",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "ICCV 2025 Open Access Repository",
     "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Zhang_VLABench_A_Large-Scale_Benchmark_for_Language-Conditioned_Robotics_Manipulation_with_Long-Horizon_ICCV_2025_paper.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "OpenMOSS/VLABench on GitHub (file README.md)",
     "url": "https://raw.githubusercontent.com/OpenMOSS/VLABench/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "OpenMOSS/VLABench on GitHub (file track_1_in_distribution.json)",
     "url": "https://raw.githubusercontent.com/OpenMOSS/VLABench/main/VLABench/configs/evaluation/tracks/track_1_in_distribution.json",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s9": {
     "title": "OpenMOSS/VLABench on GitHub (license)",
     "url": "https://api.github.com/repos/OpenMOSS/VLABench/license",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s10": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2412.18194",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    },
    "s11": {
     "title": "OpenMOSS/VLABench on GitHub (repository)",
     "url": "https://api.github.com/repos/OpenMOSS/VLABench",
     "type": "site",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "vln-ce",
   "name": "VLN-CE",
   "full_name": "Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous Environments",
   "aliases": [
    "Vision-and-Language Navigation in Continuous Environments",
    "R2R-CE",
    "RxR-CE",
    "R2R_VLNCE",
    "RxR-Habitat",
    "VLN-CE Challenge"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Fixed splits and metrics, and until January 2026 an organiser-scored test server, for simulated agents that follow route instructions with low-level moves in 3D scans of real buildings. Ports the Room-to-Room (R2R) and Room-Across-Room (RxR) datasets.",
   "summary": {
    "text": "VLN-CE is a simulated benchmark in which an agent follows a written route instruction through a 3D scan of a real building, using small moves such as 0.25 m forward or a 15-degree turn. It ports the Room-to-Room (R2R) instructions, and later the multilingual Room-Across-Room (RxR) instructions, from a graph of fixed viewpoints into the Habitat simulator.",
    "sources": [
     "s1",
     "s2",
     "s5"
    ],
    "short": "VLN-CE is a simulated benchmark in which an agent follows written route instructions through 3D scans of real buildings, using small, robot-like moves. It is widely used to test language-guided navigation models."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s1",
      "s3",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "The paper and project site define a task with fixed splits and metrics; a public test server ran on EvalAI from January 2021."
    },
    "kind_secondary": {
     "value": [
      "challenge",
      "dataset"
     ],
     "display": "Also the VLN-CE Challenge on EvalAI (2021 to January 2026) and the RxR-Habitat Challenge at CVPR 2021, 2022 and 2023, and two episode datasets (R2R_VLNCE and RxR_VLNCE).",
     "level": "verified",
     "sources": [
      "s3",
      "s5",
      "s11",
      "s17",
      "s19"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "R2R_VLNCE_v1-3; RxR_VLNCE_v0",
     "display": "Current data: R2R_VLNCE_v1-3 (the README recommends this version) and RxR_VLNCE_v0. The code targets Habitat-Sim and Habitat-Lab 0.1.7 and Python 3.6.",
     "level": "verified",
     "sources": [
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "R2R_VLNCE_v1-3",
       "display": "Train 10,819 episodes (61 scenes); val seen 778 (53); val unseen 1,839 (11); test 3,408 (18). A preprocessed version adds 146,304 augmented episodes (envdrop).",
       "level": "verified",
       "sources": [
        "s4"
       ]
      },
      {
       "value": "RxR_VLNCE_v0",
       "display": "Train 60,300 episodes (59 scenes); val seen 6,746 (57); val unseen 11,006 (11); test-challenge 9,557 (17); roughly equal English, Hindi and Telugu.",
       "level": "verified",
       "sources": [
        "s17",
        "s5",
        "s15"
       ]
      },
      {
       "value": "Waypoint models (2021-10)",
       "display": "The repository added waypoint-based baselines that use panoramic observations; the README says these are not valid RxR-Habitat submissions.",
       "level": "verified",
       "sources": [
        "s5",
        "s8"
       ]
      }
     ],
     "short": "R2R_VLNCE v1-3 and RxR_VLNCE v0"
    },
    "publishers": {
     "value": [
      "Oregon State University",
      "Georgia Tech",
      "Facebook AI Research"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s3"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Oregon State University",
       "display": "Jacob Krantz and Stefan Lee; the corresponding organiser of the challenges.",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ]
      },
      {
       "value": "Georgia Tech",
       "display": "Erik Wijmans, Arjun Majumdar, Dhruv Batra",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Facebook AI Research",
       "display": "Second affiliation of Erik Wijmans and Dhruv Batra",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ],
     "note": "The RxR data come from Google Research. The RxR-Habitat Challenge was hosted by Oregon State University, Google Research and Meta AI (README)."
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Led from Oregon State University with Georgia Tech and FAIR co-authors."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "first_release": {
     "value": "2020-04",
     "display": "arXiv v1 on 2020-04-06; code released in April 2020. Published at ECCV 2020.",
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "The GitHub repository was created on 2020-04-03 (GitHub API).",
     "short": "April 2020, at ECCV 2020"
    },
    "latest_update": {
     "value": "2026-01",
     "display": "In January 2026 the organisers announced that the EvalAI test server is being sunset; the challenge end date is 2026-01-31 and the newest test entry is dated 2026-01-17. The last code commit is 2025-01-07 (RxR data link fix).",
     "level": "verified",
     "sources": [
      "s11",
      "s13",
      "s12",
      "s8"
     ],
     "checked": "2026-10-10",
     "short": "January 2026, when the test server was retired"
    },
    "status": {
     "value": "dormant",
     "display": "Code unchanged since 2025-01-07 and the test server retired in January 2026. Use in new papers is very active.",
     "level": "inferred",
     "sources": [
      "s8",
      "s11",
      "s7",
      "s34"
     ],
     "checked": "2026-10-10",
     "note": "29 issues are open (GitHub API). The RxR repository was archived on 2026-04-19 (s34). For current use see facts.used_by.",
     "short": "No code changes since January 2025. Use is very active."
    },
    "capability": {
     "value": [
      "instruction-following",
      "navigation"
     ],
     "level": "verified",
     "sources": [
      "s1",
      "s3"
     ],
     "checked": "2026-10-10"
    },
    "generalisation": {
     "value": [
      "scene-layout",
      "language"
     ],
     "display": "Val-unseen (11 scenes) and test (18 scenes) use buildings not seen in training, with new instructions. RxR adds Hindi and Telugu instructions.",
     "level": "inferred",
     "sources": [
      "s4",
      "s17"
     ],
     "checked": "2026-10-10",
     "short": "New buildings and new instructions"
    },
    "venue": {
     "value": "sim",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "simulator": {
     "value": "Habitat-Sim 0.1.7",
     "display": "Habitat-Sim and Habitat-Lab 0.1.7 with Matterport3D scene meshes",
     "level": "verified",
     "sources": [
      "s5",
      "s2"
     ],
     "checked": "2026-10-10",
     "short": "Habitat-Sim 0.1.7"
    },
    "embodiment": {
     "value": [
      "mobile-base"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper models 'a ground-based, zero-turning radius robot with a single, forward-mounted RGBD camera, similar to a LoCoBot'."
    },
    "robots": {
     "value": "None (simulated LoCoBot-like agent)",
     "display": "No robot model. R2R setting: forward 0.25 m, turn 15 degrees, stop; RGB-D camera with a 90-degree field of view. RxR-Habitat setting: 30-degree turns, look up and down, 640 x 480 RGB-D.",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s10"
     ],
     "checked": "2026-10-10",
     "note": "The paper gives 256 x 256 RGB-D; the R2R config sets RGB to 224 x 224 and depth to 256 x 256."
    },
    "scene": {
     "value": [
      "home"
     ],
     "display": "90 building-scale Matterport3D scans",
     "level": "inferred",
     "sources": [
      "s2",
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "We did not check the share of non-residential buildings in Matterport3D."
    },
    "scenes": {
     "value": 90,
     "display": "90 Matterport3D scenes. R2R splits: train 61, val seen 53, val unseen 11, test 18 scenes.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s5"
     ],
     "checked": "2026-10-10",
     "short": "90 scanned buildings"
    },
    "tasks": {
     "value": 16844,
     "display": "16,844 R2R episodes (instruction plus path) over four splits. RxR_VLNCE adds 87,609 episodes.",
     "level": "inferred",
     "sources": [
      "s4",
      "s17"
     ],
     "checked": "2026-10-10",
     "note": "Sums by us: 10,819 + 778 + 1,839 + 3,408 for R2R; 60,300 + 6,746 + 11,006 + 9,557 for RxR.",
     "short": "16,844 R2R and 87,609 RxR episodes"
    },
    "demonstrations": {
     "value": 4475,
     "display": "4,475 R2R paths ported to continuous scenes (77% of R2R paths were navigable), each with about three instructions. Reference action sequences come with train and validation episodes; 146,304 augmented episodes are provided.",
     "level": "verified",
     "sources": [
      "s2",
      "s4",
      "s16",
      "s25"
     ],
     "checked": "2026-10-10",
     "note": "A ported path averages 55.88 low-level steps, against 4 to 6 hops in nav-graph R2R (paper Section 3.2). The paper found much lower scores than in nav-graph R2R: random agents reach about 3% success against 16.3% in R2R. Sim-2-Sim (ECCV 2022) transferred a nav-graph agent into VLN-CE and gained 12 points of success but did not keep its nav-graph performance.",
     "short": "4,475 paths ported from R2R"
    },
    "scoring": {
     "value": [
      "success-rate",
      "path-efficiency"
     ],
     "level": "verified",
     "sources": [
      "s11",
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "RxR-Habitat ranks by nDTW, a path-similarity score with no taxonomy value."
    },
    "metric_detail": {
     "value": "success rate and SPL",
     "display": "Success: the agent calls stop within 3 m of the goal, measured along walkable space. SPL: success weighted by path length; a successful episode scores the shortest-path length divided by the longer of the agent's path and the shortest path. Navigation error: distance left to the goal in metres. Oracle success: success if the agent had stopped at its closest point to the goal. RxR-Habitat ranks by nDTW, which scores from 0 to 1 how closely the agent's path follows the reference path.",
     "level": "verified",
     "sources": [
      "s11",
      "s17",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "EvalAI: the 3 m threshold is below the 5 m minimum start-to-goal distance in R2R.",
     "short": "Success (stopping within 3 m of the goal) and SPL (success weighted by path length)"
    },
    "trials": {
     "value": "one run per episode",
     "display": "One run per episode: 1,839 val-unseen or 3,408 test episodes. R2R episodes stop after 500 steps.",
     "level": "verified",
     "sources": [
      "s4",
      "s9"
     ],
     "checked": "2026-10-10"
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s12",
      "s27",
      "s30"
     ],
     "checked": "2026-10-10",
     "note": "The leaderboard and the papers we read report single numbers."
    },
    "evaluator": {
     "value": "both",
     "display": "Test scores were computed by the EvalAI server from submitted trajectories against hidden goals until January 2026. Validation scores in papers are self-reported, and the organisers now recommend reporting val-unseen.",
     "level": "verified",
     "sources": [
      "s11",
      "s12"
     ],
     "checked": "2026-10-10",
     "note": "The agent runs on the team's machine; the server only scores the trajectory file."
    },
    "leaderboard": {
     "value": "official",
     "display": "EvalAI 'VLN-CE Challenge' test leaderboard: 46 public entries from 2020-12-28 to 2026-01-17, now closed. The RxR-Habitat leaderboard sits on a Google page that we could not read.",
     "level": "verified",
     "sources": [
      "s12",
      "s11",
      "s13"
     ],
     "checked": "2026-10-10",
     "short": "Official, on EvalAI. It closed in January 2026."
    },
    "top_score": {
     "value": 66.4,
     "display": "66.4% success, SPL 0.577 on the hidden test set (VLN-CLASH, 2025-06-21). Values below are test success rates from the official leaderboard; self-reported val-unseen results, listed after them, reached 72.1% in 2026 and are not directly comparable.",
     "level": "verified",
     "sources": [
      "s12"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": 27.61,
       "display": "CMA_PM_DA_Aug baseline, 2020-12: 27.61% (SPL 0.2529)",
       "level": "verified",
       "sources": [
        "s12",
        "s5"
       ],
       "data": {
        "model": "CMA_PM_DA_Aug (baseline)",
        "date": "2020-12",
        "avg": 27.61,
        "rl": false
       }
      },
      {
       "value": 31.75,
       "display": "HPN+DN (WaypointTeam), 2021-03: 31.75%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "HPN+DN",
        "date": "2021-03",
        "avg": 31.75,
        "rl": false
       }
      },
      {
       "value": 41.93,
       "display": "CWP-VLNBERT, 2021-10: 41.93%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "CWP-VLNBERT",
        "date": "2021-10",
        "avg": 41.93,
        "rl": false
       }
      },
      {
       "value": 49.3,
       "display": "Reborn, 2022-04: 49.30%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "Reborn",
        "date": "2022-04",
        "avg": 49.3,
        "rl": false
       }
      },
      {
       "value": 55.13,
       "display": "ETPNav (TPAMI 2024), 2023-02: 55.13%. Three later reproductions scored 54.11% to 55.37%.",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "ETPNav",
        "date": "2023-02",
        "avg": 55.13,
        "rl": false
       }
      },
      {
       "value": 58.54,
       "display": "BEVBert (ICCV 2023), 2023-05: 58.54%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "BEVBert",
        "date": "2023-05",
        "avg": 58.54,
        "rl": false
       }
      },
      {
       "value": 63.91,
       "display": "SRVLN, 2024-10: 63.91%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "SRVLN",
        "date": "2024-10",
        "avg": 63.91,
        "rl": false
       }
      },
      {
       "value": 45.1,
       "display": "NaVid (RGB video only), 2024-12: 45.10%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "NaVid",
        "date": "2024-12",
        "avg": 45.1,
        "rl": false
       }
      },
      {
       "value": 66.4,
       "display": "VLN-CLASH, 2025-06: 66.40%. Highest on the leaderboard.",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "VLN-CLASH",
        "date": "2025-06",
        "avg": 66.4,
        "rl": false
       }
      },
      {
       "value": 63.67,
       "display": "ETP-R1, 2025-08: 63.67%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "ETP-R1",
        "date": "2025-08",
        "avg": 63.67,
        "rl": false
       }
      },
      {
       "value": 62.32,
       "display": "CoMaR (g3D-LF), 2025-09: 62.32%",
       "level": "verified",
       "sources": [
        "s12"
       ],
       "data": {
        "model": "CoMaR (g3D-LF)",
        "date": "2025-09",
        "avg": 62.32,
        "rl": false
       }
      },
      {
       "value": "Val-unseen, single RGB camera",
       "display": "NaVid 37.4% (2024-02); NaVILA 54.0% (2024-12); LightNav-0 68.5% (2026-08)",
       "level": "verified",
       "sources": [
        "s26",
        "s27",
        "s32"
       ]
      },
      {
       "value": "Val-unseen, several cameras",
       "display": "Qwen-RobotNav-8B 72.1%, SPL 66.6 (panoramic RGB, 2026-06); ABot-N1 70.9%, SPL 67.5 (three RGB views, 2026-07)",
       "level": "verified",
       "sources": [
        "s30",
        "s31"
       ]
      },
      {
       "value": "RxR-Habitat challenge",
       "display": "2021: no team beat the CMA baseline (nDTW 0.3086). 2022: Reborn, nDTW 0.5543. 2023 winner: The GridMM Team (score not found).",
       "level": "verified",
       "sources": [
        "s17",
        "s18",
        "s19"
       ]
      }
     ],
     "note": "Test rows are organiser-scored. The leaderboard mixes single-camera and panoramic, waypoint-based methods (see issues.i1). The 'rl' flag is false for all rows: RL, where used, happens in training scenes, which differ from test scenes.",
     "short": "66.4% success on the test set (June 2025)"
    },
    "license_code": {
     "value": "MIT",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: MIT, copyright 2020 the five authors."
    },
    "license_data": {
     "value": "CC-BY-NC-SA-3.0-US",
     "display": "VLN-CE episode datasets and trained models: CC BY-NC-SA 3.0 US plus the Matterport3D Terms of Use. RxR instruction annotations: CC BY 4.0. Original R2R data: Matterport3D Terms of Use.",
     "level": "verified",
     "sources": [
      "s5",
      "s14",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "VLN-CE README: task datasets and trained models 'are considered data derived from the mp3d scene dataset'. The Matterport3D Simulator README puts R2R under the Matterport3D Terms of Use.",
     "short": "CC BY-NC-SA 3.0 US, plus the Matterport3D terms"
    },
    "license_assets": {
     "value": "Matterport academic licence (custom)",
     "display": "Matterport3D scenes: Matterport's End User License Agreement for Academic Use. Non-commercial academic use only; models trained on the data count as derived information and may not be used for non-academic purposes.",
     "level": "verified",
     "sources": [
      "s20",
      "s21"
     ],
     "checked": "2026-10-10",
     "short": "Matterport terms for academic use only"
    },
    "access": {
     "value": "application",
     "level": "inferred",
     "sources": [
      "s5",
      "s21"
     ],
     "checked": "2026-10-10",
     "note": "Scenes: sign the Matterport3D terms and email them to receive a download script. Episode files and code are open downloads (Google Drive and GitHub)."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s5",
      "s20"
     ],
     "checked": "2026-10-10",
     "note": "Episode data CC BY-NC-SA 3.0 US and scenes under Matterport's academic-only licence, which also covers trained models. Code MIT. Not legal advice."
    },
    "sim_to_real": {
     "value": "demonstrated",
     "display": "Several VLN-CE-trained agents have been run on real robots, but no study compares VLN-CE scores and real-robot results for the same set of policies.",
     "level": "inferred",
     "sources": [
      "s24",
      "s27",
      "s26",
      "s23"
     ],
     "checked": "2026-10-10",
     "note": "VLN-PE (ICCV 2025) ran VLN-CE models with physically simulated humanoid, quadruped and wheeled robots in Isaac Sim and found a 34% relative drop in success; on a real Unitree Go2 over 14 episodes, a CMA model trained only on VLN-CE reached 7.14% success (28.57% after VLN-PE fine-tuning). NaVILA tested on a Unitree Go2 and a Booster T1 with 25 instructions repeated three times, comparing against GPT-4o, which it did not score on VLN-CE. NaVid also reports real-robot tests. Anderson et al. (CoRL 2020) ran a nav-graph R2R agent on a TurtleBot2 in an office with a scanned replica (46.8% real against 55.9% in simulation with a prepared map; 22.5% without), but that agent was trained on discrete R2R, not VLN-CE. See searched.",
     "short": "Agents trained on it have run on real robots. No study compares its scores with real-robot results."
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Simulation only."
    },
    "derived_benchmarks": {
     "value": [
      "VLN-PE",
      "VLN-CE-Isaac"
     ],
     "display": "Benchmarks that re-run VLN-CE episodes with physically simulated robots",
     "level": "verified",
     "sources": [
      "s24",
      "s27"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "VLN-PE",
       "display": "2025-07, ICCV 2025. R2R episodes with humanoid, quadruped and wheeled robots in Isaac Sim.",
       "level": "verified",
       "sources": [
        "s24"
       ]
      },
      {
       "value": "VLN-CE-Isaac",
       "display": "2024-12. Introduced with NaVILA; VLN-CE episodes in Isaac Sim with low-level control.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      }
     ],
     "note": "RxR-CE (RxR-Habitat) is part of this entry. Not a complete list.",
     "short": "2 re-runs with physically simulated robots"
    },
    "citations": {
     "value": 683,
     "display": "683 (Semantic Scholar; 148 influential)",
     "level": "verified",
     "sources": [
      "s35"
     ],
     "checked": "2026-10-10",
     "items": [],
     "short": "683"
    },
    "github_stars": {
     "value": 886,
     "display": "886 stars, 91 forks (jacobkrantz/VLN-CE)",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "short": "886"
    },
    "used_by": {
     "value": "46 public entries on the test leaderboard (2020-12 to 2026-01). Every 2025–2026 navigation model report we read reports R2R-CE val-unseen.",
     "level": "verified",
     "sources": [
      "s12",
      "s28",
      "s30",
      "s31",
      "s32"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "ETPNav, BEVBert",
       "display": "Waypoint-based methods with test-server entries, 2023",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "NaVid",
       "display": "Peking University, BAAI, Galbot and others, RSS 2024. Single RGB camera.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "NaVILA",
       "display": "UC San Diego, USC and NVIDIA, 2024-12. Legged robots.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "StreamVLN",
       "display": "Shanghai AI Lab and partners, 2025-07, ICRA 2026",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "Qwen-RobotNav",
       "display": "Qwen Team (Alibaba), 2026-06",
       "level": "verified",
       "sources": [
        "s30"
       ]
      },
      {
       "value": "ABot-N1",
       "display": "AMAP CV Lab (Alibaba Group), 2026-07",
       "level": "verified",
       "sources": [
        "s31"
       ]
      },
      {
       "value": "LightNav-0",
       "display": "Light Origins Team, 2026-08",
       "level": "verified",
       "sources": [
        "s32"
       ]
      }
     ],
     "short": "Reported by most navigation models from 2025 to 2026"
    },
    "industry_use": {
     "value": [
      "Alibaba",
      "NVIDIA",
      "Galbot",
      "Google",
      "Meta"
     ],
     "level": "verified",
     "sources": [
      "s30",
      "s31",
      "s27",
      "s26",
      "s5"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Alibaba",
       "display": "Qwen Team (Qwen-RobotNav) and AMAP CV Lab (ABot-N0, ABot-N1) report VLN-CE results.",
       "level": "verified",
       "sources": [
        "s30",
        "s31"
       ]
      },
      {
       "value": "NVIDIA",
       "display": "Co-authors of NaVILA.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "Galbot",
       "display": "Affiliation of NaVid co-authors.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "Google",
       "display": "Google Research built RxR and co-hosted the RxR-Habitat Challenge.",
       "level": "verified",
       "sources": [
        "s14",
        "s5"
       ]
      },
      {
       "value": "Meta",
       "display": "FAIR co-authored VLN-CE; Meta AI co-hosted the RxR-Habitat Challenge.",
       "level": "verified",
       "sources": [
        "s2",
        "s5"
       ]
      }
     ]
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "inconsistent-reporting",
     "title": "Methods with different sensors and data share one ranking",
     "text": "Some methods use a panoramic RGB-D camera, odometry and a waypoint predictor trained in the simulator; others use one forward RGB camera. Many recent models also train on extra VLN data beyond R2R-CE and RxR-CE (marked with a dagger in StreamVLN's table). The NaVILA, StreamVLN and ABot-N1 tables label these inputs, but the EvalAI leaderboard ranks all entries together. The RxR-Habitat rules required one 640 x 480 RGB-D camera, which excluded panoramic waypoint models.",
     "level": "verified",
     "sources": [
      "s27",
      "s28",
      "s31",
      "s12",
      "s5"
     ],
     "status": "open",
     "short": "Methods that use a single camera and methods that use a panoramic camera are ranked together."
    },
    {
     "id": "i2",
     "type": "inconsistent-reporting",
     "title": "The same model appears with different scores",
     "text": "NaVid reports 37.4% val-unseen success; Qwen-RobotNav's table lists NaVid at 41.9%. StreamVLN with extra data reported 56.9% in v1 (2025-07) and 56.4% in v2 (2026-07); later papers still cite 56.9%. ABot-N1 reports 70.9% (multi-task) and 68.3% (single-task); LightNav-0 lists ABot-N1 at 68.3%. The CMA baseline was published at 0.30 SPL on val-unseen but scores 0.27 on the leaderboard, which the README attributes to hardware and Habitat build differences.",
     "level": "verified",
     "sources": [
      "s26",
      "s30",
      "s29",
      "s28",
      "s31",
      "s32",
      "s5"
     ],
     "status": "open",
     "short": "Papers quote NaVid's success on the val-unseen split (buildings not seen in training) as 37.4% and 41.9%."
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "Results now come from a split with public answers",
     "text": "In January 2026 the organisers retired the EvalAI test server and recommended reporting val-unseen, following RxR's test leaderboard. Ground-truth paths for val-unseen ship with the data (the test split's goals were hidden). The same split can therefore be used for model selection and for the reported score.",
     "level": "verified",
     "sources": [
      "s11",
      "s4"
     ],
     "status": "open",
     "note": "The model-selection risk is our inference; we did not find a study that measures it for VLN-CE.",
     "short": "The test server was retired in January 2026. New scores come from the val-unseen split, whose answers are public."
    },
    {
     "id": "i4",
     "type": "other",
     "title": "The movement is idealised",
     "text": "VLN-PE found that VLN-CE models lose 34% of their success, in relative terms, when they must move as physically simulated robots. The default R2R-CE configuration lets the agent slide along walls on collision (ALLOW_SLIDING: True), while the RxR-Habitat configuration disables sliding. In Habitat PointNav, Kadian et al. showed that sliding inflated simulated success; whether it does so in VLN-CE has not been measured.",
     "level": "verified",
     "sources": [
      "s24",
      "s9",
      "s10",
      "s33"
     ],
     "status": "open",
     "note": "The link between sliding and VLN-CE scores is our inference.",
     "short": "In VLN-PE, agents lost about a third of their success when they had to move as physically simulated robots."
    },
    {
     "id": "i5",
     "type": "shortcut",
     "title": "Early baselines did nearly as well without the instruction",
     "text": "In the original paper, a model given no instruction reached 17% val-unseen success and a model given no RGB image also reached 17%, against 20% for the full sequence-to-sequence baseline. The authors read this as shared path regularities between R2R and VLN-CE. Later models score far higher, and we found no newer ablation of this kind on VLN-CE.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "status": "open",
     "short": "In the 2020 paper, a model given no instruction reached 17% success, against 20% for the full model."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "A VLN-CE score shows how well an agent follows route instructions through 3D scans of real buildings with small, robot-like moves. It is closer to a real robot than the original Room-to-Room (R2R) task, which moves between fixed viewpoints on a graph. Its motion is still idealised, and no study has paired VLN-CE scores with real-robot results.",
     "basis": [
      "facts.sim_to_real",
      "issues.i4",
      "facts.metric_detail"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "VLN-CE is closer to real robots than R2R. It has not been checked against real-robot results."
    },
    {
     "id": "r2",
     "text": "Before comparing two VLN-CE numbers, check the split, the camera setup (single RGB, panoramic, depth, odometry, waypoint predictor), the extra training data and the paper version. Published tables mix all of these.",
     "basis": [
      "issues.i1",
      "issues.i2",
      "facts.evaluator"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the split, sensors and training data before comparing scores."
    },
    {
     "id": "r3",
     "text": "Since January 2026, new results are self-reported on val-unseen, whose answers are public. Gains of one or two points should be read with care.",
     "basis": [
      "issues.i3",
      "facts.leaderboard"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Read small gains on val-unseen with care."
    },
    {
     "id": "r4",
     "text": "VLN-CE is the common benchmark for language-guided navigation models in 2025 and 2026, including company models from Alibaba. Its Matterport licence limits commercial use of models trained on it.",
     "basis": [
      "facts.used_by",
      "facts.industry_use",
      "facts.commercial_use"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "VLN-CE is widely used. Models trained on it are limited to academic use."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "How well an agent will do on a real robot.",
     "sub": "We found no study that compares VLN-CE scores with real-robot results for the same agents.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "How well an agent handles physical motion and collisions.",
     "sub": "The moves are idealised. In the default R2R setup, the agent slides along walls when it hits them.",
     "basis": [
      "issues.i4"
     ]
    },
    {
     "id": "l3",
     "text": "Whether a score can be fairly compared with scores in other papers.",
     "sub": "Sensors, training data and paper versions differ between papers.",
     "basis": [
      "issues.i1",
      "issues.i2"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "VLN-CE paper and project site; Sim-2-Sim (2204.09667); Anderson et al. CoRL 2020 (discrete R2R agent); VLN-PE (2507.13019, Isaac Sim plus 14 real episodes); NaVid, NaVILA, StreamVLN real-robot sections; web searches on 2026-10-10 for VLN-CE sim-to-real correlation and real-robot evaluation. No study pairs VLN-CE scores and real-robot results across several policies.",
     "date": "2026-10-10"
    },
    {
     "for": "RxR-Habitat leaderboard",
     "where": "ai.google.com/research/rxr/habitat (JavaScript page; no readable content via curl or fetch). Results taken from the organisers' retrospective, the 2022 winner report and the CVPR 2023 workshop page.",
     "date": "2026-10-10"
    },
    {
     "for": "license_data (R2R annotations)",
     "where": "bringmeaspoon.org (no licence statement); Matterport3D Simulator README (data derived from Matterport3D under its Terms of Use).",
     "date": "2026-10-10"
    },
    {
     "for": "citations",
     "where": "Semantic Scholar API, repeated HTTP 429 responses on 2026-10-10.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous Environments (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2004.02857",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2020-04",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "VLN-CE paper, full text v2 (Sections 3 to 5, Tables 2 to 4)",
     "url": "https://arxiv.org/pdf/2004.02857",
     "type": "paper",
     "publisher": "arXiv (ECCV 2020 version)",
     "date": "2020-05",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "VLN-CE project site (news, people, leaderboard link)",
     "url": "https://jacobkrantz.github.io/vlnce/",
     "type": "site",
     "publisher": "Oregon State University (Jacob Krantz)",
     "date": "2021-01",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "VLN-CE dataset page (R2R_VLNCE_v1-3 split counts and format)",
     "url": "https://jacobkrantz.github.io/vlnce/data",
     "type": "site",
     "publisher": "Oregon State University (Jacob Krantz)",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "VLN-CE GitHub README (data, RxR-Habitat Challenge, baseline performance, licence)",
     "url": "https://github.com/jacobkrantz/VLN-CE/blob/master/README.md",
     "type": "repo",
     "publisher": "jacobkrantz",
     "date": "2025-01",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "VLN-CE LICENSE file",
     "url": "https://github.com/jacobkrantz/VLN-CE/blob/master/LICENSE",
     "type": "repo",
     "publisher": "jacobkrantz",
     "date": "2020",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "GitHub API: jacobkrantz/VLN-CE (stars, forks, created, pushed)",
     "url": "https://api.github.com/repos/jacobkrantz/VLN-CE",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "VLN-CE commit history",
     "url": "https://github.com/jacobkrantz/VLN-CE/commits/master",
     "type": "repo",
     "publisher": "jacobkrantz",
     "date": "2025-01-07",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "VLN-CE R2R task configuration (vlnce_task.yaml: ALLOW_SLIDING True, 500 steps, 3.0 m success)",
     "url": "https://github.com/jacobkrantz/VLN-CE/blob/master/habitat_extensions/config/vlnce_task.yaml",
     "type": "repo",
     "publisher": "jacobkrantz",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "VLN-CE RxR English task configuration (ALLOW_SLIDING False, 30-degree turns, 640 x 480 RGB-D)",
     "url": "https://github.com/jacobkrantz/VLN-CE/blob/master/habitat_extensions/config/rxr_vlnce_english_task.yaml",
     "type": "repo",
     "publisher": "jacobkrantz",
     "date": "2021",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "EvalAI: VLN-CE Challenge (challenge 719) description, evaluation and submission guidelines, with the January 2026 sunset notice",
     "url": "https://eval.ai/web/challenges/challenge-page/719",
     "type": "leaderboard",
     "publisher": "EvalAI (host team VIRL)",
     "date": "2026-01",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "EvalAI: VLN-CE Challenge test leaderboard (phase split 1966; 46 public entries)",
     "url": "https://eval.ai/api/jobs/challenge_phase_split/1966/leaderboard/",
     "type": "leaderboard",
     "publisher": "EvalAI",
     "date": "2026-01-17",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "EvalAI API: VLN-CE Challenge phases (test: 1 per day, 5 in total; end 2026-01-31)",
     "url": "https://eval.ai/api/challenges/challenge/719/challenge_phase",
     "type": "leaderboard",
     "publisher": "EvalAI",
     "date": "2026-01-31",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "Room-Across-Room (RxR) repository README and LICENSE (CC BY 4.0; archived 2026-04-19 per GitHub)",
     "url": "https://github.com/google-research-datasets/RxR",
     "type": "repo",
     "publisher": "Google Research",
     "date": "2023-07",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Room-Across-Room: Multilingual Vision-and-Language Navigation with Dense Spatiotemporal Grounding",
     "url": "https://arxiv.org/abs/2010.07954",
     "type": "paper",
     "publisher": "EMNLP 2020 (Google Research)",
     "date": "2020-10",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "Vision-and-Language Navigation: Interpreting visually-grounded navigation instructions in real environments (R2R)",
     "url": "https://arxiv.org/abs/1711.07280",
     "type": "paper",
     "publisher": "CVPR 2018",
     "date": "2017-11",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "Retrospectives on the Embodied AI Workshop (RxR-Habitat section)",
     "url": "https://arxiv.org/abs/2210.06849",
     "type": "paper",
     "publisher": "arXiv (Embodied AI Workshop organisers)",
     "date": "2022-10",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "1st Place Solutions for RxR-Habitat Vision-and-Language Navigation Competition (CVPR 2022)",
     "url": "https://arxiv.org/abs/2206.11610",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2022-06",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "Embodied AI Workshop, CVPR 2023 (challenge table with 2023 winners)",
     "url": "https://embodied-ai.org/cvpr2023/",
     "type": "site",
     "publisher": "Embodied AI Workshop organisers",
     "date": "2023-06",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Matterport End User License Agreement for Academic Use of Model Data",
     "url": "https://matterport.com/legal/matterport-end-user-license-agreement-academic-use-model-data",
     "type": "site",
     "publisher": "Matterport",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Matterport3D Terms of Use (PDF linked from the VLN-CE README)",
     "url": "http://kaldir.vc.in.tum.de/matterport/MP_TOS.pdf",
     "type": "site",
     "publisher": "Matterport",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Matterport3D Simulator README (licence: Matterport3D data and derived data under the Matterport3D Terms of Use; code MIT)",
     "url": "https://github.com/peteanderson80/Matterport3DSimulator",
     "type": "repo",
     "publisher": "Peter Anderson et al.",
     "date": "2024-07",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "Sim-to-Real Transfer for Vision-and-Language Navigation (Anderson et al., CoRL 2020)",
     "url": "https://arxiv.org/abs/2011.03807",
     "type": "paper",
     "publisher": "CoRL 2020",
     "date": "2020-11",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "Rethinking the Embodied Gap in Vision-and-Language Navigation (VLN-PE), v2",
     "url": "https://arxiv.org/abs/2507.13019",
     "type": "paper",
     "publisher": "ICCV 2025 (Tongji University, Shanghai AI Laboratory, SJTU)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Sim-2-Sim Transfer for Vision-and-Language Navigation in Continuous Environments",
     "url": "https://arxiv.org/abs/2204.09667",
     "type": "paper",
     "publisher": "ECCV 2022 (Oregon State University)",
     "date": "2022-04",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation (v7, Table I)",
     "url": "https://arxiv.org/abs/2402.15852",
     "type": "paper",
     "publisher": "RSS 2024 (Peking University, BAAI, Galbot and others)",
     "date": "2024-02",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "NaVILA: Legged Robot Vision-Language-Action Model for Navigation (Tables I and VI)",
     "url": "https://arxiv.org/abs/2412.04453",
     "type": "paper",
     "publisher": "RSS 2025 (UC San Diego, USC, NVIDIA)",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "StreamVLN: Streaming Vision-and-Language Navigation via SlowFast Context Modeling, v2 (Table I)",
     "url": "https://arxiv.org/abs/2507.05240",
     "type": "paper",
     "publisher": "ICRA 2026 (Shanghai AI Lab, HKU, Zhejiang University, SJTU)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "StreamVLN, v1 (Table I)",
     "url": "https://arxiv.org/pdf/2507.05240v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Qwen-RobotNav Technical Report (Table 1: VLN-CE val-unseen)",
     "url": "https://arxiv.org/abs/2606.18112",
     "type": "paper",
     "publisher": "arXiv (Qwen Team, Alibaba)",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "ABot-N1: Toward a General Visual Language Navigation Foundation Model (Table 1)",
     "url": "https://arxiv.org/abs/2607.10383",
     "type": "paper",
     "publisher": "arXiv (AMAP CV Lab, Alibaba Group)",
     "date": "2026-07",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "LightNav-0: Eliciting VLM Spatial Intelligence for Generalist Embodied Navigation (Table III)",
     "url": "https://arxiv.org/abs/2608.30935",
     "type": "paper",
     "publisher": "arXiv (Light Origins Team)",
     "date": "2026-08",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "Sim2Real Predictivity (Kadian et al.): wall sliding in Habitat inflated simulated PointNav success",
     "url": "https://arxiv.org/abs/1912.06321",
     "type": "paper",
     "publisher": "IEEE RA-L 2020",
     "date": "2019-12",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "GitHub GraphQL: google-research-datasets/RxR archivedAt 2026-04-19",
     "url": "https://api.github.com/repos/google-research-datasets/RxR",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "Semantic Scholar API record for arXiv:2004.02857 (VLN-CE)",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2004.02857?fields=citationCount,influentialCitationCount",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a full entry from primary sources. Covers R2R-CE and RxR-CE (RxR-Habitat). Added the January 2026 test-server sunset, the official leaderboard history (46 public entries), licences (CC BY-NC-SA 3.0 US data, Matterport academic scene licence, CC BY 4.0 RxR annotations) and five issues."
    }
   ]
  },
  {
   "id": "vsi-bench",
   "name": "VSI-Bench",
   "full_name": "Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces",
   "aliases": [
    "Thinking in Space",
    "VSIBench",
    "VSI-Bench (tiny)",
    "VSI-Bench-Debiased",
    "Visual-Spatial Intelligence Benchmark"
   ],
   "depth": "full",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "A video question set on spatial understanding for general multimodal models, built from scans of real rooms rather than for robots. Included as borderline because robot-oriented model reports (Gemini Robotics 1.5, RoboBrain 2.0, HY-Embodied-0.5) report it.",
   "summary": {
    "text": "VSI-Bench asks multimodal AI models 5,130 questions about 288 videos that walk through real rooms, such as counting objects or estimating distances. Researchers at NYU, Yale and Stanford released it in December 2024. It scores the answers; no robot moves.",
    "sources": [
     "s2",
     "s6"
    ],
    "short": "VSI-Bench is a set of 5,130 questions about videos of real rooms, used to test how well multimodal AI models (models that take in video and text) understand space. Models only answer questions, and no robot moves."
   },
   "facts": {
    "kind": {
     "value": "benchmark",
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "The paper presents VSI-Bench as a benchmark of fixed question-answer pairs with its own metrics and evaluation code.",
     "short": "A benchmark with a fixed test set"
    },
    "publishers": {
     "value": [
      "New York University",
      "Yale University",
      "Stanford University"
     ],
     "level": "verified",
     "sources": [
      "s2",
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "Authors: Jihan Yang, Shusheng Yang, Anjali W. Gupta, Rilyn Han, Li Fei-Fei, Saining Xie.",
     "items": [
      {
       "value": "New York University",
       "display": "Four of six authors, including the senior author Saining Xie.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Yale University",
       "display": "Rilyn Han.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "Stanford University",
       "display": "Li Fei-Fei.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      }
     ]
    },
    "builder_type": {
     "value": "academic",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "From author affiliations: three universities."
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "All three institutions are in the United States."
    },
    "first_release": {
     "value": "2024-12",
     "display": "arXiv v1 2024-12-18; data released on Hugging Face 2024-12-19. Accepted at CVPR 2025 as an oral paper.",
     "level": "verified",
     "sources": [
      "s1",
      "s7",
      "s4"
     ],
     "checked": "2026-10-10",
     "short": "December 2024, at CVPR 2025"
    },
    "published_at": {
     "value": "CVPR 2025",
     "display": "CVPR 2025 (oral, per the repository README; DOI 10.1109/CVPR52734.2025.00994)",
     "level": "verified",
     "sources": [
      "s4",
      "s30"
     ],
     "checked": "2026-10-10",
     "short": "CVPR 2025 (oral)"
    },
    "latest_update": {
     "value": "2026-10-08",
     "display": "Dataset card updated 2026-10-08 (debiased-subset description, lmms-eval task names, COLM 2026 citation). Last data change 2025-11-11 (VSI-Bench-Debiased added; row order changed). Code last changed 2025-08-05 (scene meta information released).",
     "level": "verified",
     "sources": [
      "s7",
      "s37"
     ],
     "checked": "2026-10-10",
     "short": "The dataset card changed in October 2026. The data last changed in November 2025."
    },
    "version": {
     "value": "full + debiased (v1)",
     "display": "Three configurations on Hugging Face: full (5,130 questions, the default), debiased (2,362 kept) and pruned (2,768 removed). Paper v2 (2025-07-02) updated open-model results to 32 frames.",
     "level": "verified",
     "sources": [
      "s6",
      "s1",
      "s7"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "VSI-Bench (tiny)",
       "display": "400 questions, 50 per task, used for the human baseline.",
       "level": "verified",
       "sources": [
        "s2"
       ]
      },
      {
       "value": "VSI-Bench-Debiased v1",
       "display": "2,362 of 5,130 questions kept by hand-tuned filters (added 2025-11-11). A 2026-08-09 note on the card says it predates the automated pruning method and is a separate evaluation set.",
       "level": "verified",
       "sources": [
        "s6",
        "s8"
       ]
      },
      {
       "value": "Row order change",
       "display": "Rows were reordered on 2025-11-11 (commit d7cb1a3); the card says to join predictions on the question id.",
       "level": "verified",
       "sources": [
        "s6",
        "s7"
       ]
      }
     ],
     "short": "The full set has 5,130 questions. A debiased subset keeps 2,362."
    },
    "status": {
     "value": "active",
     "display": "The dataset card and debiased subset were updated through 2026-10-08. The evaluation code has not changed since 2025-08-05.",
     "level": "inferred",
     "sources": [
      "s7",
      "s37"
     ],
     "checked": "2026-10-10",
     "note": "Recent changes are documentation; the last data change was 2025-11-11.",
     "short": "Active. The dataset card was updated in October 2026."
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The paper studies 'visual-spatial intelligence' and names robotics as one motivation. The taxonomy has no separate value for spatial reasoning, so it is mapped to embodied-reasoning."
    },
    "generalisation": {
     "value": [
      "none-stated"
     ],
     "level": "inferred",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "note": "A zero-shot test with no training split. Questions come from scenes in the validation splits of the source scan datasets.",
     "short": "None stated. It is a test set only."
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Models answer questions about recorded videos. Nothing is controlled."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Videos are walk-through scans of rooms. No body or robot is controlled."
    },
    "scene": {
     "value": [
      "home",
      "office-lab",
      "industrial"
     ],
     "display": "Homes, offices, labs and factories",
     "level": "verified",
     "sources": [
      "s2",
      "s39"
     ],
     "checked": "2026-10-10",
     "note": "The paper lists residential spaces, professional settings (offices, labs) and industrial spaces (factories). A maintainer said the factory scene comes from ScanNet++ (issue #30)."
    },
    "tasks": {
     "value": 5130,
     "display": "5,130 questions in 8 task types: object count 565, absolute distance 834, object size 953, room size 288, relative distance 710, relative direction 968 (easy 217, medium 378, hard 373), route plan 194, appearance order 618.",
     "level": "verified",
     "sources": [
      "s6",
      "s9"
     ],
     "checked": "2026-10-10",
     "note": "Counts from the Hugging Face dataset statistics for the full configuration (2026-10-10). The paper says 'over 5,000'. Four task types take numerical answers (2,640 questions) and four are multiple choice (2,490). Only route-plan questions were written by people; the rest come from templates.",
     "short": "5,130 questions in 8 task types"
    },
    "scenes": {
     "value": 288,
     "display": "288 room videos from validation splits: ScanNet 88, ARKitScenes 150, ScanNet++ 50",
     "level": "verified",
     "sources": [
      "s2",
      "s9",
      "s38"
     ],
     "checked": "2026-10-10",
     "note": "288 is stated in the paper. The per-source counts are ours, from scene-name formats in the dataset statistics (inferred). Questions per source: ScanNet 2,071, ARKitScenes 1,601, ScanNet++ 1,458. The video download holds 512 videos, of which 288 are used (maintainer, issue #41).",
     "short": "288 room videos"
    },
    "scale": {
     "value": "about 2 to 3 minutes per video",
     "display": "Videos average about 2 to 3 minutes (maintainer, issue #28). The three video archives total about 5.73 GB.",
     "level": "verified",
     "sources": [
      "s10",
      "s11"
     ],
     "checked": "2026-10-10",
     "note": "5,728,450,973 bytes summed by us from the Hugging Face file listing (arkitscenes.zip, scannet.zip, scannetpp.zip).",
     "short": "Videos of about 2 to 3 minutes"
    },
    "demonstrations": {
     "value": "none",
     "display": "No training split. Training sets built with the same pipeline from the source datasets' training splits exist elsewhere: VSI-Train-10k (TsT paper) and VSI-590K (Cambrian-S).",
     "level": "verified",
     "sources": [
      "s6",
      "s8",
      "s12"
     ],
     "checked": "2026-10-10",
     "short": "None. Training sets built with the same templates exist elsewhere."
    },
    "scoring": {
     "value": [
      "accuracy"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Multiple-choice questions use accuracy. Numerical questions use Mean Relative Accuracy (MRA), a partial-credit score; the taxonomy has no value for it."
    },
    "metric_detail": {
     "value": "average of 8 task scores",
     "display": "Each task is scored on its own: multiple-choice questions by exact-match accuracy, numerical answers by MRA, the share of ten tolerance levels (from 50% down to 5% relative error) that the answer falls within. The headline is the plain average over the eight tasks (lmms-eval's default). The TsT paper averages over questions instead; the dataset card warns the two can differ.",
     "level": "verified",
     "sources": [
      "s2",
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "Average of 8 task scores"
    },
    "trials": {
     "value": "one pass, greedy decoding",
     "display": "Each question is asked once with greedy decoding (temperature 0). In paper v2 the frames differ by model: 16 for GPT-4o, 32 for open models, and the whole video for Gemini, which samples 1 frame per second. Later papers use 16 to 128 frames.",
     "level": "verified",
     "sources": [
      "s2",
      "s14",
      "s3",
      "s15"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Paper v1 to v2",
       "display": "InternVL2 models moved from 8 to 32 frames between versions; InternVL2-8B went from 34.6 to 37.5.",
       "level": "verified",
       "sources": [
        "s3",
        "s2"
       ]
      },
      {
       "value": "Frame sensitivity",
       "display": "SenseNova-SI paper: Cambrian-S-7B scores 58.6, 63.6, 66.4 and 67.5 with 16, 32, 64 and 128 frames.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      }
     ],
     "short": "One pass. The number of video frames varies by paper."
    },
    "uncertainty_reported": {
     "value": "no",
     "level": "inferred",
     "sources": [
      "s2",
      "s10",
      "s8"
     ],
     "checked": "2026-10-10",
     "note": "Papers report single scores. The maintainers ran closed models several times and say overall scores were stable but individual answers were not (issue #28). The TsT paper reports rerun ranges only for its own probe (42.6 to 42.8)."
    },
    "evaluator": {
     "value": "self-reported",
     "level": "inferred",
     "sources": [
      "s2",
      "s16"
     ],
     "checked": "2026-10-10",
     "note": "Each paper runs its own evaluation. The EASI community board runs models under one protocol."
    },
    "human_baseline": {
     "value": 79.2,
     "display": "79.2% average on VSI-Bench (tiny), 400 questions (50 per task). Evaluators could rewatch the video with no time limit. By task: object count 94.3, absolute distance 47.0, object size 60.4, room size 45.9, relative distance 94.7, relative direction 95.8, route plan 95.8, appearance order 100.0.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "The number of human evaluators is not stated in the text we read. With 50 questions per task, each task score has wide sampling error (about ±14 points at 50%, our calculation).",
     "short": "Humans scored 79.2% on a 400-question subset"
    },
    "leaderboard": {
     "value": "community",
     "display": "No official leaderboard. The EASI board (EvolvingLMMs-Lab) lists VSI-Bench and VSI-Bench-Debiased scores for 45 models, last updated 2026-07-01.",
     "level": "inferred",
     "sources": [
      "s16",
      "s17",
      "s13",
      "s4",
      "s41"
     ],
     "checked": "2026-10-10",
     "note": "The project page, GitHub README and dataset card have no leaderboard. Embodied Arena also includes VSI-Bench (arXiv 2509.15273), but its site could not be read on 2026-10-10.",
     "short": "No official leaderboard. A community board (EASI) lists scores."
    },
    "top_score": {
     "value": 73.2,
     "display": "73.2% by SpaceMind++ (May 2026). Highest result we found; humans scored 79.2% on the 400-question subset. Frame counts and averaging differ between the rows below.",
     "level": "inferred",
     "sources": [
      "s18"
     ],
     "checked": "2026-10-10",
     "note": "Each number is verified at its source; 'highest' is our judgement from the papers and boards we opened. The EASI community board lists GeoThinker at 72.6 (2026-07). Several 2025 and 2026 models were trained on data built from the same scan datasets with VSI-Bench-style templates (issues.i4).",
     "items": [
      {
       "value": 45.4,
       "display": "Gemini-1.5 Pro, 2024-12, paper. Best model at release.",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "data": {
        "model": "Gemini-1.5 Pro",
        "date": "2024-12",
        "avg": 45.4,
        "rl": false
       }
      },
      {
       "value": 69.5,
       "display": "InternVL3.5-241B-A28B (Shanghai AI Laboratory), 2025-08. The same report gives GPT-5 37.5 in its own run.",
       "level": "verified",
       "sources": [
        "s19"
       ],
       "data": {
        "model": "InternVL3.5-241B-A28B",
        "date": "2025-08",
        "avg": 69.5,
        "rl": false
       }
      },
      {
       "value": 52.9,
       "display": "GPT-5, 2025-10, run by Google for Gemini Robotics 1.5. Same table: Gemini 2.5 Pro 51.1, Gemini Robotics-ER 1.5 (thinking) 45.8.",
       "level": "verified",
       "sources": [
        "s20"
       ],
       "data": {
        "model": "GPT-5 (run by Google)",
        "date": "2025-10",
        "avg": 52.9,
        "rl": false
       }
      },
      {
       "value": 60,
       "display": "Qwen3-VL-235B-A22B (Qwen Team), 2025-11.",
       "level": "verified",
       "sources": [
        "s21"
       ],
       "data": {
        "model": "Qwen3-VL-235B-A22B",
        "date": "2025-11",
        "avg": 60,
        "rl": false
       }
      },
      {
       "value": 67.5,
       "display": "Cambrian-S-7B, 2025-11, with 128 frames (VSI-Bench-Debiased: 59.9). Re-run at 32 frames by Ouroboros-Spatial: 62.9.",
       "level": "verified",
       "sources": [
        "s12",
        "s22"
       ],
       "data": {
        "model": "Cambrian-S-7B",
        "date": "2025-11",
        "avg": 67.5,
        "rl": false
       }
      },
      {
       "value": 68.8,
       "display": "SenseNova-SI-1.1-InternVL3-8B (SenseTime Research and NTU), first posted 2025-11; 68.8 is the v4 figure (2026-03) with 64 frames. Its 1.2 checkpoint card (2025-12) gives 69.6.",
       "level": "verified",
       "sources": [
        "s15",
        "s24"
       ],
       "data": {
        "model": "SenseNova-SI (8B)",
        "date": "2025-11",
        "avg": 68.8,
        "rl": false
       }
      },
      {
       "value": 69.6,
       "display": "SpaceMind, 2025-11 (its own paper). SpaceMind++ lists it at 70.2.",
       "level": "verified",
       "sources": [
        "s23",
        "s18"
       ],
       "data": {
        "model": "SpaceMind",
        "date": "2025-11",
        "avg": 69.6,
        "rl": false
       }
      },
      {
       "value": 56,
       "display": "Gemini 3 Pro, run by the Spatial-TTT authors, 2026-03. HY-Embodied-0.5's API run gave 57.9.",
       "level": "verified",
       "sources": [
        "s25",
        "s26"
       ],
       "data": {
        "model": "Gemini 3 Pro",
        "date": "2026-03",
        "avg": 56,
        "rl": false
       }
      },
      {
       "value": 68.3,
       "display": "HY-Embodied-0.5 MoE-A32B (Tencent Robotics X and HY Vision Team), 2026-04.",
       "level": "verified",
       "sources": [
        "s26"
       ],
       "data": {
        "model": "HY-Embodied-0.5 MoE-A32B",
        "date": "2026-04",
        "avg": 68.3,
        "rl": false
       }
      },
      {
       "value": 73.2,
       "display": "SpaceMind++, 2026-05. Trained on about 900K spatial questions that include VSI-590K.",
       "level": "verified",
       "sources": [
        "s18"
       ],
       "data": {
        "model": "SpaceMind++",
        "date": "2026-05",
        "avg": 73.2,
        "rl": false
       }
      }
     ],
     "short": "73.2% (May 2026). Humans scored 79.2%.",
     "chart": {
      "max": 100,
      "unit": "",
      "label": "Average of the 8 task scores"
     }
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s5",
      "s31"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: Apache License 2.0. GitHub reports Apache-2.0.",
     "short": "Apache-2.0"
    },
    "license_data": {
     "value": "Apache-2.0",
     "display": "The dataset card says Apache-2.0. The videos come from scan datasets with non-commercial terms (see asset licence).",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10",
     "short": "Apache-2.0, as labelled on the dataset card"
    },
    "license_assets": {
     "value": [
      "ScanNet Terms of Use",
      "ScanNet++ Terms of Use",
      "ARKitScenes licence (Apple)"
     ],
     "display": "The videos come from ScanNet, ScanNet++ and ARKitScenes validation scans, each under its own terms.",
     "level": "verified",
     "sources": [
      "s27",
      "s28",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "The Hugging Face repository serves the videos without a gate under the Apache-2.0 label. The card does not discuss the source terms. Not legal advice.",
     "items": [
      {
       "value": "ScanNet Terms of Use",
       "display": "Use only for non-commercial research and educational purposes; access after signing an agreement.",
       "level": "verified",
       "sources": [
        "s27"
       ]
      },
      {
       "value": "ScanNet++ Terms of Use",
       "display": "Non-commercial research and educational purposes only; commercial use strictly prohibited; data may be shared with colleagues only after they agree to the terms.",
       "level": "verified",
       "sources": [
        "s28"
       ]
      },
      {
       "value": "ARKitScenes licence (Apple)",
       "display": "A non-commercial licence, plus commercial terms for licensees whose products had fewer than 700 million monthly active users before August 2024; larger ones must ask Apple.",
       "level": "verified",
       "sources": [
        "s29"
       ]
      }
     ],
     "short": "The source scans are under non-commercial terms"
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s6",
      "s32"
     ],
     "checked": "2026-10-10",
     "note": "Ungated Hugging Face dataset; no registration. The source scan datasets themselves require signed agreements.",
     "short": "Open. The data is on Hugging Face."
    },
    "commercial_use": {
     "value": "non-commercial",
     "level": "inferred",
     "sources": [
      "s6",
      "s27",
      "s28",
      "s29"
     ],
     "checked": "2026-10-10",
     "note": "The card says Apache-2.0, but the videos derive from ScanNet and ScanNet++, whose terms allow only non-commercial research and educational use. Not legal advice."
    },
    "sim_to_real": {
     "value": "none-found",
     "display": "No published study compares VSI-Bench scores with robot task success.",
     "level": "inferred",
     "sources": [
      "s2",
      "s34",
      "s40"
     ],
     "checked": "2026-10-10",
     "note": "MV-RoboBench (ICLR 2026) reports that strong scores on general single-view spatial benchmarks do not reliably carry over to robotic spatial questions in its own set; that compares question sets, not robot runs. Vlaser (ICLR 2026) found that gains on embodied-reasoning benchmarks, VSI-Bench among them, did not carry over to closed-loop robot control in simulation.",
     "short": "No study found"
    },
    "real_reproducibility": {
     "value": "not-applicable",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Offline question set; no robot runs."
    },
    "citations": {
     "value": 814,
     "display": "814 (Semantic Scholar; 143 influential)",
     "level": "verified",
     "sources": [
      "s30"
     ],
     "checked": "2026-10-10",
     "short": "814"
    },
    "github_stars": {
     "value": 746,
     "display": "746 stars, 47 forks (vision-x-nyu/thinking-in-space)",
     "level": "verified",
     "sources": [
      "s31"
     ],
     "checked": "2026-10-10",
     "short": "746"
    },
    "dataset_downloads": {
     "value": 9432,
     "display": "9,432 (Hub 'downloads' field), 158,229 all time, 71 likes: nyu-visionx/VSI-Bench",
     "level": "verified",
     "sources": [
      "s32"
     ],
     "checked": "2026-10-10",
     "note": "Read from the Hugging Face Hub API on 2026-10-10. We did not check the time window behind the 'downloads' field.",
     "short": "9,432 on Hugging Face"
    },
    "used_by": {
     "value": "At least 11 model reports and papers published VSI-Bench scores between 2025-07 and 2026-06 (our count of reports we opened).",
     "level": "inferred",
     "sources": [
      "s20",
      "s33",
      "s19",
      "s21",
      "s12",
      "s15",
      "s34",
      "s25",
      "s26",
      "s18",
      "s22"
     ],
     "checked": "2026-10-10",
     "note": "A lower bound. The lmms-eval toolkit ships vsibench and vsibench_debiased tasks (dataset card).",
     "items": [
      {
       "value": "RoboBrain 2.0",
       "display": "BAAI RoboBrain Team, 2025-07 (RoboBrain-32B-2.0: 42.69).",
       "level": "verified",
       "sources": [
        "s33"
       ]
      },
      {
       "value": "InternVL3.5",
       "display": "Shanghai AI Laboratory, 2025-08.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "Gemini Robotics 1.5",
       "display": "Google DeepMind, 2025-10. One of 15 benchmarks in its embodied-reasoning score.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Vlaser",
       "display": "2025-10, ICLR 2026 (Vlaser-8B: 60.3).",
       "level": "verified",
       "sources": [
        "s34"
       ]
      },
      {
       "value": "Qwen3-VL",
       "display": "Qwen Team, 2025-11.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Cambrian-S",
       "display": "2025-11. Builds VSI-590K and VSI-SUPER.",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "SenseNova-SI",
       "display": "SenseTime Research and NTU, 2025-11; CVPR 2026.",
       "level": "verified",
       "sources": [
        "s15"
       ]
      },
      {
       "value": "Spatial-TTT",
       "display": "2026-03.",
       "level": "verified",
       "sources": [
        "s25"
       ]
      },
      {
       "value": "HY-Embodied-0.5",
       "display": "Tencent Robotics X and HY Vision Team, 2026-04.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "SpaceMind++",
       "display": "2026-05.",
       "level": "verified",
       "sources": [
        "s18"
       ]
      },
      {
       "value": "Ouroboros-Spatial",
       "display": "2026-06.",
       "level": "verified",
       "sources": [
        "s22"
       ]
      }
     ],
     "short": "At least 11 reports (July 2025 to June 2026)"
    },
    "industry_use": {
     "value": [
      "Google DeepMind",
      "Alibaba (Qwen Team)",
      "Shanghai AI Laboratory",
      "Tencent",
      "SenseTime",
      "BAAI"
     ],
     "level": "verified",
     "sources": [
      "s20",
      "s21",
      "s19",
      "s26",
      "s15",
      "s33"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "Google DeepMind",
       "display": "Gemini Robotics 1.5 report.",
       "level": "verified",
       "sources": [
        "s20"
       ]
      },
      {
       "value": "Alibaba (Qwen Team)",
       "display": "Qwen3-VL report.",
       "level": "verified",
       "sources": [
        "s21"
       ]
      },
      {
       "value": "Shanghai AI Laboratory",
       "display": "InternVL3.5 report.",
       "level": "verified",
       "sources": [
        "s19"
       ]
      },
      {
       "value": "Tencent",
       "display": "HY-Embodied-0.5 report.",
       "level": "verified",
       "sources": [
        "s26"
       ]
      },
      {
       "value": "SenseTime",
       "display": "SenseNova-SI paper and model cards.",
       "level": "verified",
       "sources": [
        "s15",
        "s24"
       ]
      },
      {
       "value": "BAAI",
       "display": "RoboBrain 2.0 report.",
       "level": "verified",
       "sources": [
        "s33"
       ]
      }
     ]
    },
    "derived_benchmarks": {
     "value": [
      "VSI-Bench-Debiased",
      "VSI-SUPER",
      "ReVSI"
     ],
     "level": "verified",
     "sources": [
      "s8",
      "s12",
      "s35"
     ],
     "checked": "2026-10-10",
     "items": [
      {
       "value": "VSI-Bench-Debiased",
       "display": "2025-11, same lab. 2,362 questions kept after removing many that a text-only model can answer; COLM 2026 paper.",
       "level": "verified",
       "sources": [
        "s8",
        "s6"
       ]
      },
      {
       "value": "VSI-SUPER",
       "display": "2025-11, in the Cambrian-S paper. Its counting part joins VSI-Bench room-tour clips into videos of 10 to 120 minutes.",
       "level": "verified",
       "sources": [
        "s12"
       ]
      },
      {
       "value": "ReVSI",
       "display": "2026-04, ICML 2026 (Simon Fraser University and others). Re-annotates 381 scenes from 5 datasets and regenerates all questions to fix VSI-Bench's annotation and frame problems.",
       "level": "verified",
       "sources": [
        "s35"
       ]
      }
     ],
     "short": "3 follow-up benchmarks"
    }
   },
   "issues": [
    {
     "id": "i1",
     "type": "shortcut",
     "title": "Many questions can be answered without the video",
     "text": "The original paper found blind models below chance overall, but object-size questions answerable above chance from common knowledge. The TsT study (COLM 2026, same lab) trained a text-only Qwen2-7B on VSI-Bench's own questions by cross-validation: its score rose from 24.7 to 42.7 (+17.9) without any video. Fine-tuning LLaVA-Video-7B on VSI-Train-10k, made with VSI-Bench's templates from the source datasets' training scans, raised its score with video from 36.7 to 57.1 and its score without video from 25.9 to 44.7. So much of what such training adds can come from answer patterns.",
     "level": "verified",
     "sources": [
      "s2",
      "s8",
      "s6"
     ],
     "status": "open",
     "mitigation": {
      "text": "VSI-Bench-Debiased (2,362 questions) removes many blind-answerable questions; for the fine-tuned LLaVA-Video-7B it lowers the blind score to 32.0 and the video score to 48.7. The authors recommend reporting full and debiased scores side by side. It changes the task mix: 85.7% of medium relative-direction questions are removed against 2.3% of easy ones.",
      "sources": [
       "s8",
       "s6"
      ]
     },
     "short": "A model that saw only the text of the questions learned their answer patterns. It gained 17.9 points without seeing any video."
    },
    {
     "id": "i2",
     "type": "other",
     "title": "Some official answers are wrong",
     "text": "VSI-Bench builds answers from the 3D annotations of the source scan datasets. A user listed counting questions whose official answers disagree with the video, for example 2 towels where 12 are visible (issue #40, 2025-09); a maintainer replied that source annotations contain errors and that filtering cannot ensure all answers are correct. ReVSI (ICML 2026) manually checked all object-counting, room-size and object-size questions. It reports notable errors and ambiguity in counting, room-size errors from noisy reconstructions, and physically implausible object sizes, and says similar errors are likely in other tasks.",
     "level": "verified",
     "sources": [
      "s36",
      "s35"
     ],
     "status": "open",
     "short": "The answers come from the labels of the source 3D scans. An audit (ReVSI) found errors in counting and size questions."
    },
    {
     "id": "i3",
     "type": "protocol-variance",
     "title": "Scores depend on how many frames a model sees",
     "text": "Models see a sample of video frames, and the number varies (8 to 128 across papers; Gemini models take the whole video at 1 frame per second). ReVSI checked how many questions keep a correct answer when frames are sampled evenly: with 16 frames, 28% of appearance-order and 54% of relative-distance questions; with 32 frames, 48% and 80%; with 64 frames, 70% and 92%. Scores move with frame count: Cambrian-S-7B scores 58.6 with 16 frames and 67.5 with 128 (SenseNova-SI paper), and Ouroboros-Spatial re-ran it at 32 frames and got 62.9. ReVSI recommends at least 64 frames.",
     "level": "verified",
     "sources": [
      "s35",
      "s15",
      "s22",
      "s2",
      "s14"
     ],
     "status": "open",
     "short": "With fewer video frames, some questions cannot be answered correctly. One model's score changed by 8.9 points with the number of frames."
    },
    {
     "id": "i4",
     "type": "contamination",
     "title": "Training data is built from the same scan datasets and templates",
     "text": "The answer key is public, with no hidden test split. Several high-scoring models were trained on spatial question sets built from the training splits of the same scan datasets with VSI-Bench-style templates: VSI-590K (Cambrian-S, which follows VSI-Bench's recipe), VSI-Train-10k (TsT) and SpaceMind++'s 900K mix, which includes VSI-590K. The papers do not report test videos in these sets, but TsT shows that same-template training raises scores even without video. No study has checked scene overlap between the test videos and these training sets.",
     "level": "inferred",
     "sources": [
      "s12",
      "s8",
      "s18",
      "s6"
     ],
     "status": "open",
     "note": "A risk, not a measured leak.",
     "short": "Top models train on questions made with the same templates from the same scan datasets."
    },
    {
     "id": "i5",
     "type": "inconsistent-reporting",
     "title": "Papers report different scores for the same model",
     "text": "GPT-5: 37.5 (InternVL3.5, VLMEvalKit), 52.9 (Gemini Robotics 1.5) and 55.0 (Spatial-TTT, Ouroboros-Spatial, SenseNova-SI model cards). Gemini 2.5 Pro: 43.4 (Vlaser), 51.1 (Gemini Robotics 1.5), 51.5 (Cambrian-S) and 53.5 (Spatial-TTT). Gemini 3 Pro: 56.0 (Spatial-TTT) and 57.9 (HY-Embodied-0.5 API run). Cambrian-S-7B: 67.5 in its paper (128 frames), 62.9 in Ouroboros-Spatial and 62.92 on the EASI board. SpaceMind: 69.6 in its own paper and 70.2 in SpaceMind++. Papers also differ in averaging (by task or by question), as the dataset card warns.",
     "level": "verified",
     "sources": [
      "s19",
      "s20",
      "s25",
      "s22",
      "s24",
      "s34",
      "s12",
      "s26",
      "s16",
      "s23",
      "s18",
      "s6"
     ],
     "status": "open",
     "short": "Different papers report GPT-5's score as 37.5, 52.9 and 55.0."
    },
    {
     "id": "i6",
     "type": "other",
     "title": "The videos come with non-commercial terms from their sources",
     "text": "The Hugging Face card labels the dataset Apache-2.0 and serves the video files without a gate. The videos derive from ScanNet and ScanNet++, whose terms allow only non-commercial research and educational use after signing an agreement, and from ARKitScenes under Apple's licence. The card does not discuss these terms.",
     "level": "inferred",
     "sources": [
      "s6",
      "s27",
      "s28",
      "s29"
     ],
     "status": "open",
     "note": "Our reading of the licences. Not legal advice.",
     "short": "The dataset is labelled Apache-2.0, but the source scans allow only non-commercial use."
    }
   ],
   "readings": [
    {
     "id": "r1",
     "text": "VSI-Bench measures how well a model answers questions about the layout of real rooms from video. No study links its scores to robot task success, and in one study (Vlaser) gains on such benchmarks did not carry over to robot control in simulation.",
     "basis": [
      "facts.sim_to_real",
      "facts.venue"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "VSI-Bench measures how well a model answers spatial questions from video. No study links its scores to robot success."
    },
    {
     "id": "r2",
     "text": "Before comparing two VSI-Bench numbers, check the frame count, the averaging method and whether the model was trained on same-template data. Each of these has moved scores by between 5 and 18 points.",
     "basis": [
      "issues.i1",
      "issues.i3",
      "issues.i4",
      "issues.i5"
     ],
     "confidence": "high",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Check the frame count, the averaging and the training data before comparing scores."
    },
    {
     "id": "r3",
     "text": "The human score of 79.2 is an uneven reference. Humans scored 94 to 100 on layout and order tasks but 45.9 to 60.4 on size and distance estimates, where several 2026 models already score higher. It also rests on 50 questions per task.",
     "basis": [
      "facts.human_baseline",
      "facts.top_score"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Humans score higher than models on layout questions, but several models beat them on size and distance."
    },
    {
     "id": "r4",
     "text": "Read the debiased score next to the full score, as the authors recommend. A model whose score drops much more than others on the debiased subset is probably relying on answer patterns.",
     "basis": [
      "issues.i1",
      "facts.version"
     ],
     "confidence": "medium",
     "by": "Atlas editors",
     "date": "2026-10-10",
     "short": "Read the debiased score next to the full score."
    }
   ],
   "limits": [
    {
     "id": "l1",
     "text": "Whether a robot using the model will succeed at tasks.",
     "sub": "No study links VSI-Bench scores to robot results.",
     "basis": [
      "facts.sim_to_real"
     ]
    },
    {
     "id": "l2",
     "text": "Whether the model actually used the video.",
     "sub": "A model trained on the questions alone, without video, gained 17.9 points.",
     "basis": [
      "issues.i1"
     ]
    },
    {
     "id": "l3",
     "text": "How a score compares with scores in other papers.",
     "sub": "Papers use different numbers of video frames and different ways of averaging.",
     "basis": [
      "issues.i3",
      "issues.i5"
     ]
    }
   ],
   "validity": [],
   "searched": [
    {
     "for": "sim_to_real",
     "where": "VSI-Bench paper (v1 and v2), project page, repository issues; Gemini Robotics 1.5 report; Vlaser (2510.11027); MV-RoboBench (2510.19400); ReVSI (2604.24300); TsT (2511.04655); A2Eval (2602.01640); web searches on 2026-10-10 for studies that relate VSI-Bench to robot or VLA success. None found.",
     "date": "2026-10-10"
    },
    {
     "for": "leaderboard",
     "where": "Project page, GitHub README, Hugging Face card (none official); EASI GitHub README and board API (read); Embodied Arena paper (read) and site (script-only page, not readable).",
     "date": "2026-10-10"
    },
    {
     "for": "top_score",
     "where": "The papers and model cards under used_by and top_score; EASI board API (2026-07-01 snapshot); web searches on 2026-10-10 for VSI-Bench averages of 70 and above. SpaceMind++ (73.2) is the highest found.",
     "date": "2026-10-10"
    },
    {
     "for": "license_assets",
     "where": "Dataset card, repository LICENSE, ScanNet README and Terms of Use PDF, ScanNet++ Terms of Use PDF, ARKitScenes LICENSE and README.",
     "date": "2026-10-10"
    },
    {
     "for": "human_baseline (number of evaluators)",
     "where": "Paper Section 4.1 and Appendix C.2; not stated.",
     "date": "2026-10-10"
    },
    {
     "for": "contamination (scene overlap)",
     "where": "Cambrian-S, TsT and SpaceMind++ data sections; web search on 2026-10-10. No study checks overlap between VSI-Bench test scenes and training sets.",
     "date": "2026-10-10"
    }
   ],
   "sources": {
    "s1": {
     "title": "Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces (arXiv abstract page)",
     "url": "https://arxiv.org/abs/2412.14171",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "Thinking in Space, full text v2 (Sections 3 and 4, Figure 6, Appendix C)",
     "url": "https://arxiv.org/html/2412.14171v2",
     "type": "paper",
     "publisher": "arXiv (CVPR 2025 version)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "Thinking in Space, full text v1 (Figure 6, Table 5 frame counts)",
     "url": "https://arxiv.org/html/2412.14171v1",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "vision-x-nyu/thinking-in-space README",
     "url": "https://github.com/vision-x-nyu/thinking-in-space",
     "type": "repo",
     "publisher": "VisionX @ NYU",
     "date": "2025-08-05",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "thinking-in-space LICENSE (Apache 2.0)",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/blob/main/LICENSE",
     "type": "repo",
     "publisher": "VisionX @ NYU",
     "date": "2024-12-19",
     "accessed": "2026-10-10"
    },
    "s6": {
     "title": "nyu-visionx/VSI-Bench dataset card",
     "url": "https://huggingface.co/datasets/nyu-visionx/VSI-Bench",
     "type": "repo",
     "publisher": "NYU VisionX (Hugging Face)",
     "date": "2026-10-08",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "Hugging Face commit list for nyu-visionx/VSI-Bench",
     "url": "https://huggingface.co/api/datasets/nyu-visionx/VSI-Bench/commits/main",
     "type": "repo",
     "publisher": "NYU VisionX",
     "date": "2026-10-08",
     "accessed": "2026-10-10"
    },
    "s8": {
     "title": "Benchmark Designers Should 'Train on the Test Set' to Expose Exploitable Non-Visual Shortcuts (TsT), v2",
     "url": "https://arxiv.org/abs/2511.04655",
     "type": "paper",
     "publisher": "arXiv (COLM 2026; same lab)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s9": {
     "title": "Hugging Face dataset viewer statistics, full configuration (question types, sources, scenes)",
     "url": "https://datasets-server.huggingface.co/statistics?dataset=nyu-visionx/VSI-Bench&config=full&split=test",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s10": {
     "title": "thinking-in-space issue #28: Evaluation Frames",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/issues/28",
     "type": "repo",
     "publisher": "VisionX @ NYU (maintainer replies)",
     "date": "2025-06-27",
     "accessed": "2026-10-10"
    },
    "s11": {
     "title": "Hugging Face file listing for nyu-visionx/VSI-Bench (file sizes)",
     "url": "https://huggingface.co/api/datasets/nyu-visionx/VSI-Bench/tree/main",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s12": {
     "title": "Cambrian-S: Towards Spatial Supersensing in Video (Tables 5 and 6, VSI-590K)",
     "url": "https://arxiv.org/abs/2511.04670",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s13": {
     "title": "VSI-Bench project page",
     "url": "https://vision-x-nyu.github.io/thinking-in-space.github.io/",
     "type": "site",
     "publisher": "VisionX @ NYU",
     "date": "2024-12",
     "accessed": "2026-10-10"
    },
    "s14": {
     "title": "thinking-in-space issue #5: Number of frames used in evaluation",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/issues/5",
     "type": "repo",
     "publisher": "VisionX @ NYU (maintainer reply)",
     "date": "2025-01-08",
     "accessed": "2026-10-10"
    },
    "s15": {
     "title": "Scaling Spatial Intelligence with Multimodal Foundation Models (SenseNova-SI), v4 (abstract, Table 2)",
     "url": "https://arxiv.org/abs/2511.13719",
     "type": "paper",
     "publisher": "arXiv (SenseTime Research, NTU; CVPR 2026)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s16": {
     "title": "EASI leaderboard data API (VSI-Bench scores, 45 models)",
     "url": "https://easi.lmms-lab.com/api/leaderboard",
     "type": "leaderboard",
     "publisher": "EvolvingLMMs-Lab (community)",
     "date": "2026-07-01",
     "accessed": "2026-10-10"
    },
    "s17": {
     "title": "EvolvingLMMs-Lab/EASI README (benchmarks and leaderboard link)",
     "url": "https://github.com/EvolvingLMMs-Lab/EASI",
     "type": "repo",
     "publisher": "EvolvingLMMs-Lab",
     "date": "2026-07-01",
     "accessed": "2026-10-10"
    },
    "s18": {
     "title": "SpaceMind++: Toward Allocentric Cognitive Maps for Spatially Grounded Video MLLMs (Table 1, training data)",
     "url": "https://arxiv.org/abs/2605.09449",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-05",
     "accessed": "2026-10-10"
    },
    "s19": {
     "title": "InternVL3.5 report (Table 2 and Table 11)",
     "url": "https://arxiv.org/abs/2508.18265",
     "type": "paper",
     "publisher": "arXiv (InternVL Team, Shanghai AI Laboratory)",
     "date": "2025-08",
     "accessed": "2026-10-10"
    },
    "s20": {
     "title": "Gemini Robotics 1.5 report, full text v3 (Appendix C.1, Table 19)",
     "url": "https://arxiv.org/html/2510.03342v3",
     "type": "paper",
     "publisher": "arXiv (Gemini Robotics Team, Google DeepMind)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s21": {
     "title": "Qwen3-VL Technical Report (Section 5.8)",
     "url": "https://arxiv.org/abs/2511.21631",
     "type": "paper",
     "publisher": "arXiv (Qwen Team)",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s22": {
     "title": "Ouroboros-Spatial (Table 1: VSI-Bench at 32 frames)",
     "url": "https://arxiv.org/abs/2606.11719",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-06",
     "accessed": "2026-10-10"
    },
    "s23": {
     "title": "SpaceMind: Camera-Guided Modality Fusion for Spatial Reasoning (Table 1)",
     "url": "https://arxiv.org/abs/2511.23075",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-11",
     "accessed": "2026-10-10"
    },
    "s24": {
     "title": "SenseNova-SI-1.2-InternVL3-8B model card (VSI 69.6; GPT-5 55.0)",
     "url": "https://huggingface.co/sensenova/SenseNova-SI-1.2-InternVL3-8B",
     "type": "repo",
     "publisher": "SenseNova (Hugging Face)",
     "date": "2025-12",
     "accessed": "2026-10-10"
    },
    "s25": {
     "title": "Spatial-TTT (Table 1: VSI-Bench)",
     "url": "https://arxiv.org/abs/2603.12255",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2026-03",
     "accessed": "2026-10-10"
    },
    "s26": {
     "title": "HY-Embodied-0.5 report (Tables 1 and 2)",
     "url": "https://arxiv.org/abs/2604.07430",
     "type": "paper",
     "publisher": "arXiv (Tencent Robotics X and HY Vision Team)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s27": {
     "title": "ScanNet Terms of Use (PDF)",
     "url": "http://kaldir.vc.cit.tum.de/scannet/ScanNet_TOS.pdf",
     "type": "site",
     "publisher": "ScanNet (Stanford University, Princeton University)",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s28": {
     "title": "ScanNet++ Terms of Use (PDF)",
     "url": "https://kaldir.vc.in.tum.de/scannetpp/static/scannetpp-terms-of-use.pdf",
     "type": "site",
     "publisher": "Technical University of Munich",
     "date": "unknown",
     "accessed": "2026-10-10"
    },
    "s29": {
     "title": "ARKitScenes LICENSE (README points to it for dataset use)",
     "url": "https://github.com/apple/ARKitScenes/blob/main/LICENSE",
     "type": "repo",
     "publisher": "Apple",
     "date": "2024",
     "accessed": "2026-10-10"
    },
    "s30": {
     "title": "Semantic Scholar API record for arXiv:2412.14171",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2412.14171?fields=title,citationCount,influentialCitationCount,externalIds,publicationDate,venue",
     "type": "index",
     "publisher": "Semantic Scholar",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s31": {
     "title": "GitHub API: vision-x-nyu/thinking-in-space (stars, forks, licence)",
     "url": "https://api.github.com/repos/vision-x-nyu/thinking-in-space",
     "type": "index",
     "publisher": "GitHub",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s32": {
     "title": "Hugging Face Hub API record for nyu-visionx/VSI-Bench (downloads, likes, gating)",
     "url": "https://huggingface.co/api/datasets/nyu-visionx/VSI-Bench?expand[]=downloads&expand[]=likes&expand[]=downloadsAllTime&expand[]=gated",
     "type": "index",
     "publisher": "Hugging Face",
     "date": "2026-10-10",
     "accessed": "2026-10-10"
    },
    "s33": {
     "title": "RoboBrain 2.0 Technical Report (Table 3: VSI-Bench)",
     "url": "https://arxiv.org/abs/2507.02029",
     "type": "paper",
     "publisher": "arXiv (BAAI RoboBrain Team)",
     "date": "2025-07",
     "accessed": "2026-10-10"
    },
    "s34": {
     "title": "Vlaser (Table 1; Section 3.2 on transfer to robot control)",
     "url": "https://arxiv.org/abs/2510.11027",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s35": {
     "title": "ReVSI: Rebuilding Visual Spatial Intelligence Evaluation (Section 3, Figure 2, Appendix A Table 7)",
     "url": "https://arxiv.org/abs/2604.24300",
     "type": "paper",
     "publisher": "arXiv (ICML 2026; Simon Fraser University and others)",
     "date": "2026-04",
     "accessed": "2026-10-10"
    },
    "s36": {
     "title": "thinking-in-space issue #40: Questionable ground truth in QA and metadata",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/issues/40",
     "type": "repo",
     "publisher": "VisionX @ NYU (user report, maintainer reply)",
     "date": "2025-09-26",
     "accessed": "2026-10-10"
    },
    "s37": {
     "title": "thinking-in-space commit history",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/commits/main",
     "type": "repo",
     "publisher": "VisionX @ NYU",
     "date": "2025-08-05",
     "accessed": "2026-10-10"
    },
    "s38": {
     "title": "thinking-in-space issue #41: number of videos",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/issues/41",
     "type": "repo",
     "publisher": "VisionX @ NYU (maintainer reply)",
     "date": "2025-10-14",
     "accessed": "2026-10-10"
    },
    "s39": {
     "title": "thinking-in-space issue #30: the factory scene",
     "url": "https://github.com/vision-x-nyu/thinking-in-space/issues/30",
     "type": "repo",
     "publisher": "VisionX @ NYU (maintainer reply)",
     "date": "2025-07-21",
     "accessed": "2026-10-10"
    },
    "s40": {
     "title": "Seeing Across Views: MV-RoboBench (Sections 2.4 and 4)",
     "url": "https://arxiv.org/abs/2510.19400",
     "type": "paper",
     "publisher": "arXiv (ICLR 2026)",
     "date": "2025-10",
     "accessed": "2026-10-10"
    },
    "s41": {
     "title": "Embodied Arena: A Comprehensive, Unified, and Evolving Evaluation Platform for Embodied AI",
     "url": "https://arxiv.org/abs/2509.15273",
     "type": "paper",
     "publisher": "arXiv",
     "date": "2025-09",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a full entry from primary sources; no basic entry existed. Leads came from the niches, surveys and merged sweep records. Prior sweep claims checked: 288 videos from ScanNet, ScanNet++ and ARKitScenes validation splits, 5,130 questions, human 79% (79.2), Apache-2.0 card, 814 citations, 746 stars, 9,432 Hub downloads, debiased subset of 2,362, row-order change on 2025-11-11 and the 2026-08-09 provenance note were all confirmed. The sweep's 'best model about 33 points below humans' refers to the 2024 paper; the best result found now is 73.2. The sweep listed RoboBrain 2.0 as a user; confirmed."
    }
   ]
  },
  {
   "id": "wagibench",
   "name": "WAGIBench",
   "aliases": [
    "Wearable Agent Goal Inference Benchmark",
    "OB2"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "borderline",
   "scope_reason": "Egocentric multimodal benchmark for a wearable assistant. No body is controlled and no physical action is scored; models only infer the human's goal. Matches the borderline class 'egocentric human video benchmarks'. Leaning exclude.",
   "summary": {
    "text": "Tests whether vision-language models can guess a smart-glasses wearer's goal from video, audio and phone context.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Meta Reality Labs; Meta FAIR",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Each author is marked Meta Reality Labs or Meta FAIR (a few authors show no affiliation)."
    },
    "builder_type": {
     "value": "frontier-lab",
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "region": {
     "value": "north-america",
     "level": "inferred",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Meta Platforms principal executive offices: Menlo Park, California."
    },
    "first_release": {
     "value": "2025-10 (arXiv v1 2025-10-25; repo initial commit 2025-10-22; dataset tarball last-modified 2025-10-23)",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Repo created 2025-07-02 per GitHub API (https://api.github.com/repos/facebookresearch/WAGIBench), first commit 2025-10-22."
    },
    "latest_update": {
     "value": "2026-02: last commit 2026-02-12 (dependency version bumps); data access instructions updated 2025-10-27",
     "level": "verified",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "No dataset version change found."
    },
    "version": {
     "value": "OB2 v2.4.4 data files (e.g. ob2_v2.4.4_1_mcq.json)",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10"
    },
    "published_at": {
     "value": "NeurIPS 2025 Datasets and Benchmarks Track (spotlight per arXiv comment)",
     "level": "verified",
     "sources": [
      "s6"
     ],
     "checked": "2026-10-10"
    },
    "venue": {
     "value": "offline",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Models answer from recorded observations; no control loop."
    },
    "capability": {
     "value": [
      "embodied-reasoning"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Poor fit; see taxonomy_friction."
    },
    "embodiment": {
     "value": [
      "none"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "scene": {
     "value": [
      "mixed"
     ],
     "level": "inferred",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Scripted everyday activities (memory, learning, health and fitness, meal preparation, chores, recreation)."
    },
    "scale": {
     "checked": "2026-10-10",
     "items": [
      {
       "value": "arXiv v1: 29 hours, 348 participants, 3,477 recordings. NeurIPS version: ~30 hours, 363 participants, 3,482 recordings.",
       "level": "verified",
       "sources": [
        "s6"
       ],
       "note": "Conflict between versions; arXiv numbers from https://arxiv.org/abs/2510.22443. Raw capture 264 h, 155 h after quality review, 29 h after context windowing (arXiv v1 Appendix D)."
      },
      {
       "value": "~7k multiple-choice questions (one 'similar' and one 'dissimilar' MCQ per sample, 3 distractors each); human study subset 586 samples; dataset tarball 17,085,909,384 bytes",
       "level": "verified",
       "sources": [
        "s2"
       ],
       "note": "Tarball size from HTTP headers of https://dl.fbaipublicfiles.com/wagibench/ob2.tar.gz."
      }
     ]
    },
    "scoring": {
     "value": [
      "accuracy",
      "auto-judge"
     ],
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "note": "Paper reports bootstrapped 95% confidence intervals of the mean (Fig. 4). Self-reported. Repo reference run: Qwen2.5-VL-72B MCQ 0.871, generative 0.507 (https://github.com/facebookresearch/WAGIBench/blob/main/README.md)."
    },
    "leaderboard": {
     "value": "paper-only",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Repo gives raw predictions for paper models; no leaderboard."
    },
    "license_code": {
     "value": "Apache-2.0",
     "level": "verified",
     "sources": [
      "s7"
     ],
     "checked": "2026-10-10",
     "note": "LICENSE file: Apache License 2.0."
    },
    "license_data": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "README says 'This project is licensed under the Apache 2.0 license' but names no separate data licence; the 17 GB tarball was not downloaded to look for one. Paper says participants consented to public release."
    },
    "access": {
     "value": "open",
     "level": "verified",
     "sources": [
      "s5"
     ],
     "checked": "2026-10-10",
     "note": "Direct download link, no gating seen."
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Not applicable in practice: no robot, no physical action."
    },
    "kind": {
     "value": "benchmark",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "Benchmarking Egocentric Multimodal Goal Inference for Assistive Wearable Agents",
     "url": "https://arxiv.org/abs/2510.22443",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s2": {
     "title": "Benchmarking Egocentric Multimodal Goal Inference for Assistive Wearable Agents (full text)",
     "url": "https://arxiv.org/html/2510.22443",
     "type": "paper",
     "accessed": "2026-10-10",
     "publisher": "arXiv",
     "date": "2025-10"
    },
    "s3": {
     "title": "Document",
     "url": "https://www.sec.gov/Archives/edgar/data/1326801/000162828026025534/meta-12312025x10kars.htm",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "facebookresearch/WAGIBench on GitHub (repository)",
     "url": "https://api.github.com/repos/facebookresearch/WAGIBench",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s5": {
     "title": "facebookresearch/WAGIBench on GitHub (file README.md)",
     "url": "https://github.com/facebookresearch/WAGIBench/blob/main/README.md",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s6": {
     "title": "Benchmarking Egocentric Multimodal Goal Inference for Assistive Wearable Agents",
     "url": "https://papers.neurips.cc/paper_files/paper/2025/hash/23ab960082db936f874b171822e0d097-Abstract-Datasets_and_Benchmarks_Track.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s7": {
     "title": "facebookresearch/WAGIBench on GitHub (blob)",
     "url": "https://github.com/facebookresearch/WAGIBench/blob/main/LICENSE",
     "type": "repo",
     "accessed": "2026-10-10",
     "publisher": "GitHub"
    },
    "s8": {
     "title": "Semantic Scholar API record",
     "url": "https://api.semanticscholar.org/graph/v1/paper/arXiv:2510.22443",
     "type": "index",
     "accessed": "2026-10-10",
     "publisher": "Semantic Scholar"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  },
  {
   "id": "yd-t-6770-2026",
   "name": "YD/T 6770-2026",
   "full_name": "YD/T 6770-2026 Embodied intelligence benchmark testing methods (EAI Bench)",
   "aliases": [
    "EAI Bench",
    "EAI-bench",
    "YD/T 6770-2026",
    "人工智能 关键基础技术 具身智能基准测试方法"
   ],
   "depth": "basic",
   "last_checked": "2026-10-10",
   "in_scope": "yes",
   "scope_reason": "Formal test standard for embodied AI systems issued under MIIT; standards are explicitly in scope.",
   "summary": {
    "text": "Chinese telecom-industry standard defining how to benchmark an embodied AI system in simulation and on real hardware.",
    "sources": [
     "s1"
    ]
   },
   "facts": {
    "publishers": {
     "value": "Approved by MIIT; technical committee China Communications Standards Association (CCSA); 33 drafting units led by CAICT; 69 named drafters.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "News (CCTV, S&T Daily) says CAICT 'with 40+ units'; registry lists 33. Conflict recorded."
    },
    "first_release": {
     "value": "Issued 2026-03-11.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Public announcements came later: CAICT post 2026-03-23, CCTV 2026-03-26/27, People's Daily 2026-03-30."
    },
    "latest_update": {
     "value": "Implementation 2026-06-01; filing no. 106194-2026 dated 2026-06-19.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "version": {
     "value": "YD/T 6770-2026, newly drafted (制定), method standard; CCS L70, ICS 35.020.",
     "level": "verified",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10"
    },
    "access": {
     "value": null,
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10",
     "display": "Full text not viewable on the registry: 'not yet public', reason 'copyright'."
    },
    "venue": {
     "value": "sim+real",
     "level": "reported",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Same list in CCTV (via 21jingji) and S&T Daily (via ncsti.gov.cn)."
    },
    "scoring": {
     "value": [
      "success-rate"
     ],
     "level": "reported",
     "sources": [
      "s3"
     ],
     "checked": "2026-10-10",
     "note": "Standard text not public, so definitions unverified."
    },
    "capability": {
     "value": [],
     "level": "reported",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10",
     "note": "CCTV report via 21jingji; same in S&T Daily."
    },
    "embodiment": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "note": "Looked at registry scope, CAICT post, gov.cn, CCTV, People's Daily, Xinhua."
    },
    "scale": {
     "value": "Supporting task library of 10,000+ tasks covering 300 task types in industry, home, retail, logistics; tools for data collection, sim task generation, automatic metric calculation.",
     "level": "reported",
     "sources": [
      "s4"
     ],
     "checked": "2026-10-10"
    },
    "leaderboard": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "None (No public results list or leaderboard found.)",
     "note": "Searched CAICT news coverage, Xinhua 2026-06-01 article and web search; CAICT website not machine-readable."
    },
    "license_data": {
     "value": "Copyrighted standard text; registry withholds it for copyright reasons.",
     "level": "verified",
     "sources": [
      "s2"
     ],
     "checked": "2026-10-10"
    },
    "sim_to_real": {
     "value": null,
     "level": "unknown",
     "sources": [],
     "checked": "2026-10-10",
     "display": "Not checked (No published sim-vs-real comparison under this standard found.)",
     "note": "Standard covers both simulated and real testing; no published comparison of sim vs real results under it was found."
    },
    "scene": {
     "value": [
      "industrial",
      "home",
      "retail-logistics"
     ],
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the researcher from the primary page; no separate fact row."
    },
    "kind": {
     "value": "standard",
     "level": "inferred",
     "sources": [
      "s1"
     ],
     "checked": "2026-10-10",
     "note": "Classified by the Atlas from how the authors describe and distribute it."
    }
   },
   "issues": [],
   "readings": [],
   "sources": {
    "s1": {
     "title": "全国标准信息公共服务平台",
     "url": "https://hbba.sacinfo.org.cn/stdDetail/41709a554c33138fc0efd2309473da6e2de076132063af6ac21b236db93f51fb",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s2": {
     "title": "全国标准信息公共服务平台",
     "url": "https://hbba.sacinfo.org.cn/portal/online/41709a554c33138fc0efd2309473da6e2de076132063af6ac21b236db93f51fb",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s3": {
     "title": "具身智能标准里程碑：《YD/T 6770-2026 人工智能 关键基础技术 具身智能基准测试方法》（EAI bench）正式发布",
     "url": "https://finance.sina.com.cn/wm/2026-03-23/doc-inhryxfp3981110.shtml",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s4": {
     "title": "https://m.21jingji.com/article/20260327/herald/927fc7915a61eb07f55098f41ef037b0.html",
     "url": "https://m.21jingji.com/article/20260327/herald/927fc7915a61eb07f55098f41ef037b0.html",
     "type": "site",
     "accessed": "2026-10-10"
    },
    "s5": {
     "title": "PAGE TITLE GOES HERE",
     "url": "https://www.itu.int/ITU-T/workprog/wp_item.aspx?isn=23465",
     "type": "site",
     "accessed": "2026-10-10"
    }
   },
   "history": [
    {
     "date": "2026-10-10",
     "change": "Created as a basic entry: identity facts checked at primary sources (phase 1 re-verification)."
    }
   ]
  }
 ]
}