{
 "meta": {
  "checked": "2026-10-10"
 },
 "genealogy": {
  "nodes": [
   {
    "id": "action-conditional-atari",
    "name": "Action-conditional video prediction (Atari)",
    "org": [
     "University of Michigan"
    ],
    "date": "2015-07",
    "family": "action-video",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": null,
    "summary": "Predicts future Atari game frames from past frames and the player's action with convolutional and recurrent networks.",
    "why_it_matters": "It is one of the first models to predict long video sequences that depend on control inputs, the basic idea behind later world models.",
    "url": "https://arxiv.org/abs/1507.08750",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1507.08750",
      "title": "Action-Conditional Video Prediction using Deep Networks in Atari Games",
      "type": "paper",
      "date": "2015-07-31",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "physical-interaction-video",
    "name": "Physical interaction video prediction",
    "org": [
     "UC Berkeley",
     "OpenAI",
     "Google Brain"
    ],
    "date": "2016-05",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": null,
    "summary": "Learns from unlabelled robot pushing videos to predict how pixels move when the robot acts.",
    "why_it_matters": "It moved action-conditioned video prediction from games to real robot data.",
    "url": "https://arxiv.org/abs/1605.07157",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1605.07157",
      "title": "Unsupervised Learning for Physical Interaction through Video Prediction",
      "type": "paper",
      "date": "2016-05-23",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "deep-visual-foresight",
    "name": "Deep Visual Foresight",
    "org": [
     "Google Brain",
     "UC Berkeley"
    ],
    "date": "2016-10",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": null,
    "summary": "A real robot plans pushing motions by imagining future camera images with a learned video prediction model and choosing the actions whose predicted outcome matches the goal.",
    "why_it_matters": "It is an early example of a robot planning inside a learned video model instead of a hand-built simulator.",
    "url": "https://arxiv.org/abs/1610.00696",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1610.00696",
      "title": "Deep Visual Foresight for Planning Robot Motion",
      "type": "paper",
      "date": "2016-10-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "interaction-networks",
    "name": "Interaction Networks",
    "org": [
     "Google DeepMind"
    ],
    "date": "2016-12",
    "family": "learned-simulator",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "A neural network that reasons about objects and the relations between them to predict physical dynamics such as collisions and orbits.",
    "why_it_matters": "It started the graph-network line of learned physics simulators.",
    "url": "https://arxiv.org/abs/1612.00222",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1612.00222",
      "title": "Interaction Networks for Learning about Objects, Relations and Physics",
      "type": "paper",
      "date": "2016-12-01",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "world-models-2018",
    "name": "World Models",
    "org": [
     "Google Brain",
     "NNAISENSE",
     "IDSIA"
    ],
    "date": "2018-03",
    "family": "latent-dynamics",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": null,
    "summary": "Compresses game images into a small code, learns to predict how that code changes, and trains a simple controller, in part entirely inside the model's own generated 'dream'.",
    "why_it_matters": "It gave the field its name and the idea of training an agent inside a learned model of its environment.",
    "url": "https://arxiv.org/abs/1803.10122",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1803.10122",
      "title": "World Models",
      "type": "paper",
      "date": "2018-03-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "planet",
    "name": "PlaNet",
    "org": [
     "Google Brain",
     "University of Toronto",
     "DeepMind",
     "Google Research",
     "University of Michigan"
    ],
    "date": "2018-11",
    "family": "latent-dynamics",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "Learns environment dynamics from images and chooses actions by fast online planning in a learned latent space.",
    "why_it_matters": "Its recurrent state-space model became the core of the Dreamer agents.",
    "url": "https://arxiv.org/abs/1811.04551",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1811.04551",
      "title": "Learning Latent Dynamics for Planning from Pixels",
      "type": "paper",
      "date": "2018-11-12",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "muzero",
    "name": "MuZero",
    "org": [
     "DeepMind",
     "University College London"
    ],
    "date": "2019-11",
    "family": "latent-dynamics",
    "domains": [
     "games"
    ],
    "open_weights": null,
    "summary": "Combines tree search with a learned model that predicts reward, policy and value, and reaches superhuman play in Go, chess, shogi and Atari without being given the rules.",
    "why_it_matters": "It showed that a world model only needs to predict what matters for planning, not every pixel.",
    "url": "https://arxiv.org/abs/1911.08265",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1911.08265",
      "title": "Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model",
      "type": "paper",
      "date": "2019-11-19",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "dreamer",
    "name": "Dreamer",
    "org": [
     "University of Toronto",
     "Google Brain",
     "DeepMind"
    ],
    "date": "2019-12",
    "family": "latent-dynamics",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "Learns behaviours purely from trajectories imagined in the compact state space of a learned world model.",
    "why_it_matters": "It made learning in imagination practical for visual control tasks.",
    "url": "https://arxiv.org/abs/1912.01603",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1912.01603",
      "title": "Dream to Control: Learning Behaviors by Latent Imagination",
      "type": "paper",
      "date": "2019-12-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "dreamerv2",
    "name": "DreamerV2",
    "org": [
     "Google Research",
     "DeepMind",
     "University of Toronto"
    ],
    "date": "2020-10",
    "family": "latent-dynamics",
    "domains": [
     "games"
    ],
    "open_weights": null,
    "summary": "A Dreamer agent with discrete latent states that learns Atari games inside its world model.",
    "why_it_matters": "It was the first agent trained in a learned world model to reach human-level performance on the Atari benchmark, according to its authors.",
    "url": "https://arxiv.org/abs/2010.02193",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2010.02193",
      "title": "Mastering Atari with Discrete World Models",
      "type": "paper",
      "date": "2020-10-05",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "td-mpc",
    "name": "TD-MPC",
    "org": [
     "UC San Diego"
    ],
    "date": "2022-03",
    "family": "latent-dynamics",
    "domains": [
     "robotics",
     "research"
    ],
    "open_weights": null,
    "summary": "Learns a task-oriented latent dynamics model with a value function and plans short action sequences with it for continuous control.",
    "why_it_matters": "It started a planning-based line of latent world models used widely for simulated robot control.",
    "url": "https://arxiv.org/abs/2203.04955",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2203.04955",
      "title": "Temporal Difference Learning for Model Predictive Control",
      "type": "paper",
      "date": "2022-03-09",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "daydreamer",
    "name": "DayDreamer",
    "org": [
     "UC Berkeley"
    ],
    "date": "2022-06",
    "family": "latent-dynamics",
    "domains": [
     "robotics"
    ],
    "open_weights": null,
    "summary": "Runs the Dreamer algorithm online on four physical robots, so they learn from real experience without a simulator.",
    "why_it_matters": "It showed world-model learning working directly on real robot hardware.",
    "url": "https://arxiv.org/abs/2206.14176",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2206.14176",
      "title": "DayDreamer: World Models for Physical Robot Learning",
      "type": "paper",
      "date": "2022-06-28",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "iris",
    "name": "IRIS",
    "org": [
     "University of Geneva"
    ],
    "date": "2022-09",
    "family": "latent-dynamics",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "Turns game frames into discrete tokens, predicts them with a Transformer, and trains an agent inside this model on the Atari 100k benchmark.",
    "why_it_matters": "It brought Transformer sequence models into world-model agents.",
    "url": "https://arxiv.org/abs/2209.00588",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2209.00588",
      "title": "Transformers are Sample-Efficient World Models",
      "type": "paper",
      "date": "2022-09-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/eloialonso/iris",
      "title": "eloialonso/iris pretrained models",
      "type": "repo",
      "date": "2024-05",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "dreamerv3",
    "name": "DreamerV3",
    "org": [
     "Google DeepMind",
     "University of Toronto"
    ],
    "date": "2023-01",
    "family": "latent-dynamics",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": null,
    "summary": "One configuration of the Dreamer algorithm that works across over 150 tasks and collects diamonds in Minecraft from scratch.",
    "why_it_matters": "It made world-model reinforcement learning work across many domains without tuning.",
    "url": "https://arxiv.org/abs/2301.04104",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2301.04104",
      "title": "Mastering Diverse Domains through World Models",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "td-mpc2",
    "name": "TD-MPC2",
    "org": [
     "UC San Diego"
    ],
    "date": "2023-10",
    "family": "latent-dynamics",
    "domains": [
     "robotics",
     "research"
    ],
    "open_weights": true,
    "summary": "Improves TD-MPC so one set of settings works across many continuous-control tasks and scales to a single multi-task agent; the authors released hundreds of trained checkpoints.",
    "why_it_matters": "It is a common open baseline for latent world models in simulated robot control.",
    "url": "https://arxiv.org/abs/2310.16828",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2310.16828",
      "title": "TD-MPC2: Scalable, Robust World Models for Continuous Control",
      "type": "paper",
      "date": "2023-10-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/nicklashansen/tdmpc2",
      "title": "nicklashansen/tdmpc2 checkpoints",
      "type": "repo",
      "date": "2023-10",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "diamond",
    "name": "DIAMOND",
    "org": [
     "University of Geneva",
     "University of Edinburgh",
     "Microsoft Research"
    ],
    "date": "2024-05",
    "family": "action-video",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "Trains a reinforcement learning agent inside a diffusion model that predicts game frames in pixel space, on Atari, and also makes a playable Counter-Strike model.",
    "why_it_matters": "It brought diffusion video models into world-model agents and influenced later playable game models.",
    "url": "https://arxiv.org/abs/2405.12399",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.12399",
      "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
      "type": "paper",
      "date": "2024-05-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/eloialonso/diamond",
      "title": "eloialonso/diamond pretrained models",
      "type": "repo",
      "date": "2024-10",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "dreamer-4",
    "name": "Dreamer 4",
    "org": [
     "Google DeepMind"
    ],
    "date": "2025-09",
    "family": "latent-dynamics",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": null,
    "summary": "Trains an agent by reinforcement learning inside a fast video world model of Minecraft that learns mostly from unlabelled video and runs in real time on one GPU.",
    "why_it_matters": "It is the first agent reported to obtain diamonds in Minecraft from offline data only, a step toward training robots in imagination.",
    "url": "https://arxiv.org/abs/2509.24527",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2509.24527",
      "title": "Training Agents Inside of Scalable World Models",
      "type": "paper",
      "date": "2025-09-29",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "jepa-proposal",
    "name": "JEPA proposal",
    "org": [
     "Meta AI"
    ],
    "date": "2022-02",
    "family": "predictive-representation",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "Yann LeCun's proposed architecture for autonomous AI, with a world-model module and a Joint Embedding Predictive Architecture that predicts in an abstract representation space.",
    "why_it_matters": "It set out the research programme behind Meta's JEPA models.",
    "url": "https://ai.meta.com/blog/yann-lecun-advances-in-ai-research/",
    "level": "verified",
    "sources": [
     {
      "url": "https://ai.meta.com/blog/yann-lecun-advances-in-ai-research/",
      "title": "Yann LeCun on a vision to make AI systems learn and reason like animals and humans",
      "type": "blog",
      "date": "2022-02-23",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Date is Meta's blog of 2022-02-23. The June 2022 position paper on OpenReview could not be opened (browser check); later JEPA papers cite it."
   },
   {
    "id": "v-jepa",
    "name": "V-JEPA",
    "org": [
     "FAIR at Meta",
     "Inria",
     "École normale supérieure",
     "Université Gustave Eiffel",
     "New York University"
    ],
    "date": "2024-02",
    "family": "predictive-representation",
    "domains": [
     "general-video",
     "research"
    ],
    "open_weights": true,
    "summary": "Learns video representations only by predicting the features of masked parts of videos, without pixels, text or labels.",
    "why_it_matters": "It extended the JEPA idea from images to video, the basis of V-JEPA 2.",
    "url": "https://arxiv.org/abs/2404.08471",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2404.08471",
      "title": "Revisiting Feature Prediction for Learning Visual Representations from Video",
      "type": "paper",
      "date": "2024-02-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/facebookresearch/jepa",
      "title": "facebookresearch/jepa (README)",
      "type": "repo",
      "date": "2024-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "dino-wm",
    "name": "DINO-WM",
    "org": [
     "New York University",
     "Meta AI"
    ],
    "date": "2024-11",
    "family": "predictive-representation",
    "domains": [
     "robotics",
     "research"
    ],
    "open_weights": true,
    "summary": "Learns a world model on top of pre-trained DINOv2 image features from offline data and uses it to plan robot and navigation tasks without task-specific training.",
    "why_it_matters": "It showed that predicting in a pre-trained feature space can support zero-shot planning.",
    "url": "https://arxiv.org/abs/2411.04983",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2411.04983",
      "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
      "type": "paper",
      "date": "2024-11-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/gaoyuezhou/dino_wm",
      "title": "gaoyuezhou/dino_wm (README)",
      "type": "repo",
      "date": "2025-01",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "v-jepa-2",
    "name": "V-JEPA 2 and V-JEPA 2-AC",
    "org": [
     "FAIR at Meta",
     "Mila"
    ],
    "date": "2025-06",
    "family": "predictive-representation",
    "domains": [
     "robotics",
     "general-video"
    ],
    "open_weights": true,
    "summary": "Pre-trains a video model on over 1 million hours of internet video, then post-trains an action-conditioned version (V-JEPA 2-AC) on less than 62 hours of robot video to plan pick-and-place on real arms.",
    "why_it_matters": "It is the main open example of a world model that predicts in feature space rather than pixels and is used for real robot planning.",
    "url": "https://arxiv.org/abs/2506.09985",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09985",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/facebookresearch/vjepa2",
      "title": "facebookresearch/vjepa2 (README)",
      "type": "repo",
      "date": "2025-06-25",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Meta's repository added V-JEPA 2.1 on 2026-03-16; it is not a separate node here."
   },
   {
    "id": "gns",
    "name": "Graph Network Simulator (GNS)",
    "org": [
     "DeepMind",
     "Stanford University"
    ],
    "date": "2020-02",
    "family": "learned-simulator",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "Represents fluids, sand and other materials as particles in a graph and learns to simulate their motion.",
    "why_it_matters": "It showed learned simulators can handle many materials and long rollouts.",
    "url": "https://arxiv.org/abs/2002.09405",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2002.09405",
      "title": "Learning to Simulate Complex Physics with Graph Networks",
      "type": "paper",
      "date": "2020-02-21",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "meshgraphnets",
    "name": "MeshGraphNets",
    "org": [
     "DeepMind"
    ],
    "date": "2020-10",
    "family": "learned-simulator",
    "domains": [
     "research"
    ],
    "open_weights": null,
    "summary": "Learns mesh-based physics simulation such as cloth, structural mechanics and aerodynamics with graph networks.",
    "why_it_matters": "It extended learned simulators to mesh-based engineering physics and runs 1-2 orders of magnitude faster than the solvers it was trained on, according to its authors.",
    "url": "https://arxiv.org/abs/2010.03409",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2010.03409",
      "title": "Learning Mesh-Based Simulation with Graph Networks",
      "type": "paper",
      "date": "2020-10-07",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "nerd",
    "name": "NeRD (Neural Robot Dynamics)",
    "org": [
     "NVIDIA",
     "University of Washington"
    ],
    "date": "2025-08",
    "family": "learned-simulator",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "Learns robot-specific dynamics models for articulated robots under contact and plugs them into a simulator in place of the analytical dynamics.",
    "why_it_matters": "It applies learned simulators directly to robot bodies and policy training.",
    "url": "https://arxiv.org/abs/2508.15755",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.15755",
      "title": "Neural Robot Dynamics",
      "type": "paper",
      "date": "2025-08-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/NVlabs/neural-robot-dynamics",
      "title": "NVlabs/neural-robot-dynamics (README)",
      "type": "repo",
      "date": "2025-09",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "video-diffusion-models",
    "name": "Video Diffusion Models",
    "org": [
     "Google"
    ],
    "date": "2022-04",
    "family": "video-generation",
    "domains": [
     "general-video",
     "research"
    ],
    "open_weights": null,
    "summary": "Extends image diffusion models to generate temporally coherent video.",
    "why_it_matters": "It is a common root for the diffusion video generators later used as world simulators.",
    "url": "https://arxiv.org/abs/2204.03458",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2204.03458",
      "title": "Video Diffusion Models",
      "type": "paper",
      "date": "2022-04-07",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper lists only google.com author addresses, so the organisation is given as Google."
   },
   {
    "id": "unipi",
    "name": "UniPi",
    "org": [
     "MIT",
     "Google DeepMind",
     "UC Berkeley",
     "Georgia Tech",
     "University of Alberta"
    ],
    "date": "2023-01",
    "family": "world-action",
    "domains": [
     "robotics"
    ],
    "open_weights": null,
    "summary": "Turns a text goal into a generated video of the robot doing the task and derives actions from the video with an inverse dynamics model.",
    "why_it_matters": "It started the 'generate a video, then read off the actions' approach used by later robot world models.",
    "url": "https://arxiv.org/abs/2302.00111",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2302.00111",
      "title": "Learning Universal Policies via Text-Guided Video Generation",
      "type": "paper",
      "date": "2023-01-31",
      "accessed": "2026-10-10"
     }
    ],
    "note": "UniPi uses a separate inverse dynamics model; it is placed in world-action following the coordinator's family list."
   },
   {
    "id": "stable-video-diffusion",
    "name": "Stable Video Diffusion",
    "org": [
     "Stability AI"
    ],
    "date": "2023-11",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": true,
    "summary": "An openly released latent video diffusion model for text-to-video and image-to-video generation.",
    "why_it_matters": "Its open weights became the starting point for several driving and robot world models.",
    "url": "https://arxiv.org/abs/2311.15127",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2311.15127",
      "title": "Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets",
      "type": "paper",
      "date": "2023-11-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/stabilityai/stable-video-diffusion-img2vid",
      "title": "stabilityai/stable-video-diffusion-img2vid",
      "type": "repo",
      "date": "2023-11",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "sora",
    "name": "Sora",
    "org": [
     "OpenAI"
    ],
    "date": "2024-02",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": false,
    "summary": "A diffusion Transformer trained on video and image patches that generates up to a minute of high-fidelity video.",
    "why_it_matters": "Its report framed scaling video generators as 'a promising path' toward general simulators of the physical world.",
    "url": "https://openai.com/index/video-generation-models-as-world-simulators/",
    "level": "verified",
    "sources": [
     {
      "url": "https://openai.com/index/video-generation-models-as-world-simulators/",
      "title": "Video generation models as world simulators",
      "type": "blog",
      "date": "2024-02-15",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The report says model and implementation details are not included; no weights were released."
   },
   {
    "id": "veo-2",
    "name": "Veo 2",
    "org": [
     "Google DeepMind"
    ],
    "date": "2024-12",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": false,
    "summary": "Google's video generation model with improved handling of real-world physics and human movement, at resolutions up to 4K.",
    "why_it_matters": "It is the base model of Google DeepMind's robot policy evaluator.",
    "url": "https://blog.google/innovation-and-ai/models-and-research/google-labs/video-image-generation-update-december-2024/",
    "level": "verified",
    "sources": [
     {
      "url": "https://blog.google/innovation-and-ai/models-and-research/google-labs/video-image-generation-update-december-2024/",
      "title": "State-of-the-art video and image generation with Veo 2 and Imagen 3",
      "type": "blog",
      "date": "2024-12-16",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Available through Google products; no weights released (inferred from the announcement)."
   },
   {
    "id": "cosmos",
    "name": "Cosmos world foundation models",
    "org": [
     "NVIDIA"
    ],
    "date": "2025-01",
    "family": "video-generation",
    "domains": [
     "robotics",
     "driving"
    ],
    "open_weights": true,
    "summary": "A platform of open video world foundation models, tokenizers and data tools that developers fine-tune into world models for robots and vehicles.",
    "why_it_matters": "It made large open video world models available to robotics and driving teams and is the base of many later robot world models.",
    "url": "https://arxiv.org/abs/2501.03575",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2501.03575",
      "title": "Cosmos World Foundation Model Platform for Physical AI",
      "type": "paper",
      "date": "2025-01-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/nvidia/Cosmos-1.0-Diffusion-7B-Text2World",
      "title": "nvidia/Cosmos-1.0-Diffusion-7B-Text2World",
      "type": "repo",
      "date": "2025-01-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "wan",
    "name": "Wan",
    "org": [
     "Alibaba Group"
    ],
    "date": "2025-02",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": true,
    "summary": "An open suite of diffusion Transformer video generation models (Wan 2.1) released with weights.",
    "why_it_matters": "It is a frequent open base model for robot and interactive world models.",
    "url": "https://arxiv.org/abs/2503.20314",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.20314",
      "title": "Wan: Open and Advanced Large-Scale Video Generative Models",
      "type": "paper",
      "date": "2025-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/Wan-AI/Wan2.1-T2V-14B",
      "title": "Wan-AI/Wan2.1-T2V-14B",
      "type": "repo",
      "date": "2025-02",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Model weights appeared on Hugging Face in February 2025, before the March 2025 report."
   },
   {
    "id": "veo-3",
    "name": "Veo 3",
    "org": [
     "Google DeepMind"
    ],
    "date": "2025-05",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": false,
    "summary": "Google's video model that improves on Veo 2 and generates video with synchronised audio.",
    "why_it_matters": "Google DeepMind researchers reported that it solves many visual tasks zero-shot, which supports the view of video models as general world simulators.",
    "url": "https://blog.google/innovation-and-ai/products/generative-media-models-io-2025/",
    "level": "verified",
    "sources": [
     {
      "url": "https://blog.google/innovation-and-ai/products/generative-media-models-io-2025/",
      "title": "Fuel your creativity with new generative media models and tools",
      "type": "blog",
      "date": "2025-05-20",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2509.20328",
      "title": "Video models are zero-shot learners and reasoners",
      "type": "paper",
      "date": "2025-09-24",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "sora-2",
    "name": "Sora 2",
    "org": [
     "OpenAI"
    ],
    "date": "2025-09",
    "family": "video-generation",
    "domains": [
     "general-video"
    ],
    "open_weights": false,
    "summary": "OpenAI's video and audio generation model, released with a social app, that the company says follows physics more closely than prior systems.",
    "why_it_matters": "OpenAI presents it as a step toward world simulation.",
    "url": "https://openai.com/index/sora-2/",
    "level": "verified",
    "sources": [
     {
      "url": "https://openai.com/index/sora-2/",
      "title": "Sora 2 is here",
      "type": "blog",
      "date": "2025-09-30",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "cosmos-predict-2-5",
    "name": "Cosmos-Predict2.5",
    "org": [
     "NVIDIA"
    ],
    "date": "2025-10",
    "family": "video-generation",
    "domains": [
     "robotics",
     "driving"
    ],
    "open_weights": true,
    "summary": "Unifies text-, image- and video-to-world generation in one flow-based model trained on 200M curated video clips, released at 2B and 14B sizes.",
    "why_it_matters": "It is the base model for NVIDIA's later robot world models such as DreamDojo.",
    "url": "https://arxiv.org/abs/2511.00062",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2511.00062",
      "title": "World Simulation with Video Foundation Models for Physical AI",
      "type": "paper",
      "date": "2025-10-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/nvidia/Cosmos-Predict2.5-2B",
      "title": "nvidia/Cosmos-Predict2.5-2B",
      "type": "repo",
      "date": "2025",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "cosmos-3",
    "name": "Cosmos 3",
    "org": [
     "NVIDIA"
    ],
    "date": "2026-05",
    "family": "video-generation",
    "domains": [
     "robotics",
     "driving",
     "general-video"
    ],
    "open_weights": true,
    "summary": "A family of open omnimodal world models that process and generate language, image, video, audio and action in one mixture-of-transformers model.",
    "why_it_matters": "It merges video generation, world simulation and world-action models into one open release.",
    "url": "https://arxiv.org/abs/2606.02800",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.02800",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/nvidia/Cosmos3-Super",
      "title": "nvidia/Cosmos3-Super model card",
      "type": "repo",
      "date": "2026-05-31",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The model card gives a release date of 2026-05-31; arXiv v1 is 2026-06-01. Licence: OpenMDW 1.1."
   },
   {
    "id": "unisim",
    "name": "UniSim",
    "org": [
     "UC Berkeley",
     "Google DeepMind",
     "MIT",
     "University of Alberta"
    ],
    "date": "2023-10",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": null,
    "summary": "Learns an interactive simulator of real-world scenes from diverse video data, predicting video in response to text or low-level robot actions.",
    "why_it_matters": "It showed that policies trained in a learned video simulator can transfer to real robots.",
    "url": "https://arxiv.org/abs/2310.06114",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2310.06114",
      "title": "Learning Interactive Real-World Simulators",
      "type": "paper",
      "date": "2023-10-09",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "1x-world-model",
    "name": "1X World Model",
    "org": [
     "1X Technologies"
    ],
    "date": "2024-09",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": false,
    "summary": "Predicts future video from a humanoid robot's observations and actions, trained on thousands of hours of EVE robot data, to evaluate robot policies.",
    "why_it_matters": "It is an early company world model built for robot policy evaluation, with a public challenge and dataset.",
    "url": "https://www.1x.tech/discover/1x-world-model",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/1x-world-model",
      "title": "1X World Model",
      "type": "blog",
      "date": "2024-09-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/discover/redwood-ai-world-model",
      "title": "1X World Model (policy evaluation update)",
      "type": "blog",
      "date": "2025-06-16",
      "accessed": "2026-10-11"
     }
    ],
    "note": "1X released a dataset and challenge baseline weights, not this model. A June 2025 update reported correlation between world-model and real evaluations."
   },
   {
    "id": "dreamgen",
    "name": "DreamGen",
    "org": [
     "NVIDIA",
     "University of Washington",
     "KAIST",
     "UCLA",
     "UC San Diego",
     "Caltech",
     "NTU",
     "University of Maryland",
     "UT Austin"
    ],
    "date": "2025-05",
    "family": "video-generation",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "Fine-tunes video world models on robot data to generate videos of new tasks, labels them with pseudo-actions, and trains robot policies on these synthetic trajectories.",
    "why_it_matters": "It made video generators a source of training data for robot policies.",
    "url": "https://arxiv.org/abs/2505.12705",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.12705",
      "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
      "type": "paper",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/NVIDIA/GR00T-Dreams",
      "title": "NVIDIA/GR00T-Dreams (README)",
      "type": "repo",
      "date": "2025-06",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Open weights refers to the Cosmos-Predict2 checkpoints released with NVIDIA's GR00T-Dreams pipeline."
   },
   {
    "id": "genie-envisioner",
    "name": "Genie Envisioner",
    "org": [
     "AgiBot",
     "LV-NUS Lab",
     "Beihang University"
    ],
    "date": "2025-08",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A robot manipulation platform built on an instruction-conditioned video diffusion model (GE-Base), with an action module (GE-Act) and an action-conditioned simulator (GE-Sim).",
    "why_it_matters": "It is a leading Chinese open robot world-model platform.",
    "url": "https://arxiv.org/abs/2508.05635",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.05635",
      "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
      "type": "paper",
      "date": "2025-08-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/agibot-world/Genie-Envisioner",
      "title": "agibot-world/Genie-Envisioner",
      "type": "repo",
      "date": "2025-08",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "unifolm-wma-0",
    "name": "UnifoLM-WMA-0",
    "org": [
     "Unitree"
    ],
    "date": "2025-09",
    "family": "world-action",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A world-model-action framework whose video world model works either as a simulator that generates robot data or, with an action head, as part of the policy.",
    "why_it_matters": "It is an open robot world model from a Chinese humanoid maker.",
    "url": "https://github.com/unitreerobotics/unifolm-world-model-action",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/unitreerobotics/unifolm-world-model-action",
      "title": "unitreerobotics/unifolm-world-model-action (README)",
      "type": "repo",
      "date": "2025-09-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/unitreerobotics/UnifoLM-WMA-0",
      "title": "unitreerobotics/UnifoLM-WMA-0 model repository",
      "type": "repo",
      "date": "2025-09",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "ctrl-world",
    "name": "Ctrl-World",
    "org": [
     "Stanford University",
     "Tsinghua University"
    ],
    "date": "2025-10",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A controllable multi-view world model, initialised from Stable Video Diffusion, in which generalist robot policies can be rolled out to evaluate and improve them.",
    "why_it_matters": "It is a widely used open tool for evaluating robot policies inside a world model.",
    "url": "https://arxiv.org/abs/2510.10125",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Robert-gyj/Ctrl-World",
      "title": "Robert-gyj/Ctrl-World (README)",
      "type": "repo",
      "date": "2025-10",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "veo-robotics-sim",
    "name": "Veo world simulator for Gemini Robotics",
    "org": [
     "Google DeepMind"
    ],
    "date": "2025-12",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": false,
    "summary": "Adapts Veo to robot action conditioning and multi-view consistency to evaluate Gemini Robotics policies, including out-of-distribution and safety cases.",
    "why_it_matters": "It is a frontier lab using a video model as the evaluation environment for its own robot policies.",
    "url": "https://arxiv.org/abs/2512.10675",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.10675",
      "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
      "type": "paper",
      "date": "2025-12-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "cosmos-policy",
    "name": "Cosmos Policy",
    "org": [
     "NVIDIA",
     "Stanford University"
    ],
    "date": "2026-01",
    "family": "world-action",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "Turns the Cosmos-Predict2 video model into a robot policy that generates actions, future images and values as latent frames in one model.",
    "why_it_matters": "It shows a pretrained video model becoming a robot policy without new architecture.",
    "url": "https://arxiv.org/abs/2601.16163",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.16163",
      "title": "Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning",
      "type": "paper",
      "date": "2026-01-22",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/nvlabs/cosmos-policy",
      "title": "nvlabs/cosmos-policy (README)",
      "type": "repo",
      "date": "2025-12",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "1xwm-policy",
    "name": "1XWM as NEO policy",
    "org": [
     "1X Technologies"
    ],
    "date": "2026-01",
    "family": "world-action",
    "domains": [
     "robotics"
    ],
    "open_weights": false,
    "summary": "Generates a video of the NEO humanoid doing a task from a text prompt with a 14B video model, then extracts the actions with an inverse dynamics model.",
    "why_it_matters": "It is a company deploying a video world model as the robot's policy.",
    "url": "https://www.1x.tech/discover/world-model-self-learning",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "title": "1X World Model | From Video to Action: A New Way Robots Learn",
      "type": "blog",
      "date": "2026-01-12",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Video and actions come from two models (world model plus inverse dynamics model)."
   },
   {
    "id": "dreamdojo",
    "name": "DreamDojo",
    "org": [
     "NVIDIA",
     "HKUST",
     "UC Berkeley",
     "University of Washington",
     "Stanford University",
     "KAIST"
    ],
    "date": "2026-02",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A robot world model pre-trained on 44k hours of egocentric human video with continuous latent actions, then post-trained on robot data and distilled to 10.81 FPS.",
    "why_it_matters": "It uses human video at scale to teach a robot world model, for teleoperation, policy evaluation and planning.",
    "url": "https://arxiv.org/abs/2602.06949",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/NVIDIA/DreamDojo",
      "title": "NVIDIA/DreamDojo (README)",
      "type": "repo",
      "date": "2026-02-18",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "dreamzero",
    "name": "DreamZero",
    "org": [
     "NVIDIA"
    ],
    "date": "2026-02",
    "family": "world-action",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A 14B World Action Model on a pretrained video diffusion backbone that predicts future video and robot actions together and runs closed-loop control at 7Hz.",
    "why_it_matters": "Its authors report over 2x better generalisation to new tasks and environments than leading vision-language-action models in real robot tests.",
    "url": "https://arxiv.org/abs/2602.15922",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.15922",
      "title": "World Action Models are Zero-shot Policies",
      "type": "paper",
      "date": "2026-02-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/dreamzero0/dreamzero",
      "title": "dreamzero0/dreamzero (README)",
      "type": "repo",
      "date": "2026-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "ge-sim-2",
    "name": "GE-Sim 2.0",
    "org": [
     "AgiBot",
     "Beihang University",
     "LV-NUS Lab",
     "Tianjin University"
    ],
    "date": "2026-05",
    "family": "action-video",
    "domains": [
     "robotics"
    ],
    "open_weights": true,
    "summary": "A closed-loop multi-view video world simulator for robot manipulation, built on Genie Envisioner and Cosmos-Predict2, used to evaluate and train policies.",
    "why_it_matters": "It continues the Chinese Genie Envisioner line, and its model card reports a win in the WorldArena CVPR challenge.",
    "url": "https://arxiv.org/abs/2605.27491",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2605.27491",
      "title": "GE-Sim 2.0: A Roadmap Towards Comprehensive Closed-loop Video World Simulators for Robotic Manipulation",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/agibot-world/Genie-Envisioner-Sim-v2.0",
      "title": "agibot-world/Genie-Envisioner-Sim-v2.0 model card",
      "type": "repo",
      "date": "2026-05",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "genie",
    "name": "Genie",
    "org": [
     "Google DeepMind",
     "University of British Columbia"
    ],
    "date": "2024-02",
    "family": "action-video",
    "domains": [
     "games"
    ],
    "open_weights": false,
    "summary": "An 11B-parameter model trained on 30,000 hours of unlabelled 2D platformer gameplay video that learns latent actions so users can play generated worlds frame by frame.",
    "why_it_matters": "It showed that controllable worlds can be learned from video without action labels.",
    "url": "https://arxiv.org/abs/2402.15391",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "date": "2024-02-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper states the authors chose not to release the trained checkpoints."
   },
   {
    "id": "gamengen",
    "name": "GameNGen",
    "org": [
     "Google Research",
     "Tel Aviv University",
     "Google DeepMind"
    ],
    "date": "2024-08",
    "family": "interactive-world",
    "domains": [
     "games"
    ],
    "open_weights": null,
    "summary": "A diffusion model that simulates the game DOOM interactively in real time.",
    "why_it_matters": "It showed a neural model can stand in for a game engine at interactive speed.",
    "url": "https://arxiv.org/abs/2408.14837",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2408.14837",
      "title": "Diffusion Models Are Real-Time Game Engines",
      "type": "paper",
      "date": "2024-08-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "oasis",
    "name": "Oasis",
    "org": [
     "Decart",
     "Etched"
    ],
    "date": "2024-10",
    "family": "interactive-world",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "A Transformer world model that generates a playable Minecraft-like world in real time at 20 frames per second from keyboard input; a 500M-parameter version was released.",
    "why_it_matters": "It was the first openly released real-time playable world model.",
    "url": "https://oasis-model.github.io/",
    "level": "verified",
    "sources": [
     {
      "url": "https://oasis-model.github.io/",
      "title": "Oasis: A Universe in a Transformer",
      "type": "site",
      "date": "2024-10-31",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/Etched/oasis-500m",
      "title": "Etched/oasis-500m model repository",
      "type": "repo",
      "date": "2024-10-31",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "genie-2",
    "name": "Genie 2",
    "org": [
     "Google DeepMind"
    ],
    "date": "2024-12",
    "family": "action-video",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": false,
    "summary": "An autoregressive latent diffusion model that turns a single image into a playable 3D environment, consistent for up to a minute.",
    "why_it_matters": "It moved Genie from 2D games to 3D worlds for training and evaluating agents.",
    "url": "https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/",
    "level": "verified",
    "sources": [
     {
      "url": "https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/",
      "title": "Genie 2: A large-scale foundation world model",
      "type": "blog",
      "date": "2024-12-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The blog says a distilled version runs in real time with lower quality. No weights released."
   },
   {
    "id": "muse-wham",
    "name": "Muse (WHAM)",
    "org": [
     "Microsoft Research",
     "Ninja Theory"
    ],
    "date": "2025-02",
    "family": "action-video",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "A World and Human Action Model trained on about 500,000 gameplay sessions of Bleeding Edge that generates game visuals, controller actions or both.",
    "why_it_matters": "It is an open game world model aimed at game designers.",
    "url": "https://www.nature.com/articles/s41586-025-08600-3",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.nature.com/articles/s41586-025-08600-3",
      "title": "World and Human Action Models towards gameplay ideation (Nature)",
      "type": "paper",
      "date": "2025-02-19",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.microsoft.com/en-us/research/blog/introducing-muse-our-first-generative-ai-model-designed-for-gameplay-ideation/",
      "title": "Introducing Muse: Our first generative AI model designed for gameplay ideation",
      "type": "blog",
      "date": "2025-02-19",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/microsoft/wham",
      "title": "microsoft/wham model repository",
      "type": "repo",
      "date": "2025-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "odyssey-1",
    "name": "Odyssey-1",
    "org": [
     "Odyssey"
    ],
    "date": "2025-05",
    "family": "interactive-world",
    "domains": [
     "general-video"
    ],
    "open_weights": false,
    "summary": "A real-time playable world model that streams a new realistic video frame every 40 ms in response to user input.",
    "why_it_matters": "It brought interactive world models to real-world video scenes.",
    "url": "https://odyssey.systems/introducing-odyssey-1",
    "level": "verified",
    "sources": [
     {
      "url": "https://odyssey.systems/introducing-odyssey-1",
      "title": "Introducing Odyssey-1: A Playable World Model",
      "type": "blog",
      "date": "2025-05-28",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Offered as a research preview and API; no weights found."
   },
   {
    "id": "genie-3",
    "name": "Genie 3",
    "org": [
     "Google DeepMind"
    ],
    "date": "2025-08",
    "family": "interactive-world",
    "domains": [
     "games",
     "research"
    ],
    "open_weights": false,
    "summary": "Generates worlds from a text prompt that users navigate in real time at 24 frames per second and 720p, consistent for a few minutes, with promptable world events.",
    "why_it_matters": "It is Google DeepMind's first real-time world model and the base of Project Genie (January 2026) and the Waymo World Model.",
    "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
    "level": "verified",
    "sources": [
     {
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "title": "Genie 3: A new frontier for world models",
      "type": "blog",
      "date": "2025-08-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/project-genie/",
      "title": "Project Genie: AI world model now available for Ultra users in U.S.",
      "type": "blog",
      "date": "2026-01-29",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "matrix-game-2",
    "name": "Matrix-Game 2.0",
    "org": [
     "Skywork AI"
    ],
    "date": "2025-08",
    "family": "interactive-world",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "An open real-time interactive world model that generates long game-like videos with few-step autoregressive diffusion.",
    "why_it_matters": "It is an open Chinese alternative to closed real-time world models.",
    "url": "https://arxiv.org/abs/2508.13009",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.13009",
      "title": "Matrix-game 2.0: An open-source, real-time, and streaming interactive world model",
      "type": "paper",
      "date": "2025-08-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/Skywork/Matrix-Game-2.0",
      "title": "Skywork/Matrix-Game-2.0",
      "type": "repo",
      "date": "2025-08",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "title": "SkyworkAI/Matrix-Game (README news)",
      "type": "repo",
      "date": "2026-03-27",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The repository lists the release as 2025-08-12; arXiv v1 is 2025-08-18."
   },
   {
    "id": "gwm-1",
    "name": "GWM-1",
    "org": [
     "Runway"
    ],
    "date": "2025-12",
    "family": "interactive-world",
    "domains": [
     "general-video",
     "robotics"
    ],
    "open_weights": false,
    "summary": "An autoregressive model built on Runway's Gen-4.5 that generates video frame by frame in real time under camera, robot or audio control, in three variants: Worlds, Avatars and Robotics.",
    "why_it_matters": "It is a creative-video company entering robot simulation.",
    "url": "https://runwayml.com/research/introducing-runway-gwm-1",
    "level": "verified",
    "sources": [
     {
      "url": "https://runwayml.com/research/introducing-runway-gwm-1",
      "title": "Introducing Runway GWM-1",
      "type": "blog",
      "date": "2025-12-11",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Access through Runway's products and SDK; no weights found."
   },
   {
    "id": "hy-world-1-5",
    "name": "HY-World 1.5 (WorldPlay)",
    "org": [
     "Tencent Hunyuan",
     "HKUST",
     "Beihang University"
    ],
    "date": "2025-12",
    "family": "interactive-world",
    "domains": [
     "games",
     "3d"
    ],
    "open_weights": true,
    "summary": "An open streaming video diffusion model that generates 720p video at 24 FPS under keyboard and mouse control while keeping long-term geometric consistency.",
    "why_it_matters": "It adds real-time interaction to Tencent's 3D world line.",
    "url": "https://arxiv.org/abs/2512.14614",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.14614",
      "title": "WorldPlay: Towards Long-Term Geometric Consistency for Real-Time Interactive World Modeling",
      "type": "paper",
      "date": "2025-12-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://3d-models.hunyuan.tencent.com/world/world1_5/HYWorld_1.5_Tech_Report.pdf",
      "title": "HY-World 1.5 Technical Report",
      "type": "paper",
      "date": "2025-12-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/HY-WorldPlay",
      "title": "Tencent-Hunyuan/HY-WorldPlay (README)",
      "type": "repo",
      "date": "2025-12-17",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "lingbot-world",
    "name": "LingBot-World",
    "org": [
     "Robbyant (Ant Group)"
    ],
    "date": "2026-01",
    "family": "interactive-world",
    "domains": [
     "games",
     "robotics"
    ],
    "open_weights": true,
    "summary": "An open world simulator built from the Wan2.2 video model that responds in under 1 second while producing 16 frames per second, with minute-level consistency.",
    "why_it_matters": "It is an open real-time world model from Ant Group's embodied AI unit.",
    "url": "https://arxiv.org/abs/2601.20540",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.20540",
      "title": "Advancing Open-source World Models",
      "type": "paper",
      "date": "2026-01-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/robbyant/lingbot-world",
      "title": "robbyant/lingbot-world (README)",
      "type": "repo",
      "date": "2026-01-29",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "matrix-game-3",
    "name": "Matrix-Game 3.0",
    "org": [
     "Skywork AI"
    ],
    "date": "2026-03",
    "family": "interactive-world",
    "domains": [
     "games"
    ],
    "open_weights": true,
    "summary": "Adds long-horizon memory to Matrix-Game 2.0 for real-time 720p interactive video, trained on Unreal Engine, game and real-world data.",
    "why_it_matters": "It is the newest open release in a leading Chinese interactive world-model series.",
    "url": "https://arxiv.org/abs/2604.08995",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.08995",
      "title": "Matrix-Game 3.0: Real-Time and Streaming Interactive World Model with Long-Horizon Memory",
      "type": "paper",
      "date": "2026-04-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "title": "SkyworkAI/Matrix-Game (README news)",
      "type": "repo",
      "date": "2026-03-27",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The repository lists the release as 2026-03-27; arXiv v1 is 2026-04-10."
   },
   {
    "id": "oasis-3",
    "name": "Oasis 3",
    "org": [
     "Decart"
    ],
    "date": "2026-06",
    "family": "interactive-world",
    "domains": [
     "driving",
     "robotics"
    ],
    "open_weights": false,
    "summary": "A real-time action-conditioned world model with three synchronised camera views, sold through an API, starting with autonomous-vehicle training.",
    "why_it_matters": "It moves a gaming world model toward physical AI.",
    "url": "https://decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
    "level": "verified",
    "sources": [
     {
      "url": "https://decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
      "title": "Introducing Oasis 3: First Interactive World Model for Physical AI",
      "type": "blog",
      "date": "2026-06-10",
      "accessed": "2026-10-11"
     }
    ],
    "note": "API access only; no weights found."
   },
   {
    "id": "odyssey-3",
    "name": "Odyssey-3",
    "org": [
     "Odyssey"
    ],
    "date": "2026-09",
    "family": "world-action",
    "domains": [
     "robotics",
     "driving",
     "games"
    ],
    "open_weights": false,
    "summary": "An autoregressive diffusion Transformer world model whose representations drive policies for robot arms, humanoids, cars, drones and video games with a few hours of task data.",
    "why_it_matters": "It is a startup positioning one world model as the base for many physical systems.",
    "url": "https://odyssey.systems/introducing-odyssey-3",
    "level": "verified",
    "sources": [
     {
      "url": "https://odyssey.systems/introducing-odyssey-3",
      "title": "Introducing Odyssey-3: A General-Purpose Physical Intelligence",
      "type": "blog",
      "date": "2026-09-15",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Announced 2026-09-15 with public release promised 'in the coming weeks'; no weights found."
   },
   {
    "id": "hunyuanworld-1",
    "name": "HunyuanWorld 1.0",
    "org": [
     "Tencent Hunyuan"
    ],
    "date": "2025-07",
    "family": "3d-world",
    "domains": [
     "3d",
     "games"
    ],
    "open_weights": true,
    "summary": "Generates explorable 3D worlds from text or images using panoramic world proxies, separable objects and exportable 3D meshes.",
    "why_it_matters": "It is an open 3D world generator from a large Chinese lab.",
    "url": "https://arxiv.org/abs/2507.21809",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2507.21809",
      "title": "HunyuanWorld 1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/tencent/HunyuanWorld-1",
      "title": "tencent/HunyuanWorld-1",
      "type": "repo",
      "date": "2025-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "marble",
    "name": "Marble",
    "org": [
     "World Labs"
    ],
    "date": "2025-09",
    "family": "3d-world",
    "domains": [
     "3d"
    ],
    "open_weights": false,
    "summary": "Creates persistent 3D worlds from text, images, video or coarse 3D layouts that can be edited and exported as Gaussian splats, meshes or video.",
    "why_it_matters": "It is the main commercial 3D world model product.",
    "url": "https://www.worldlabs.ai/blog/marble-world-model",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.worldlabs.ai/blog/bigger-better-worlds",
      "title": "Generating Bigger and Better Worlds",
      "type": "blog",
      "date": "2025-09-16",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.worldlabs.ai/blog/marble-world-model",
      "title": "Marble: A Multimodal World Model",
      "type": "blog",
      "date": "2025-11-12",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Limited beta preview announced 2025-09-16; general availability 2025-11-12."
   },
   {
    "id": "atlas",
    "name": "Atlas",
    "org": [
     "World Labs"
    ],
    "date": "2026-09",
    "family": "3d-world",
    "domains": [
     "3d",
     "robotics"
    ],
    "open_weights": false,
    "summary": "An omni world model trained from scratch on text, images, video and 3D that generates camera-controlled video, reconstructs scenes in 3D and supports real-to-sim workflows for robots.",
    "why_it_matters": "It is World Labs' next-generation model, released shortly before the company agreed to join AMD.",
    "url": "https://www.worldlabs.ai/blog/atlas",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.worldlabs.ai/blog/atlas",
      "title": "Atlas: A World Model for Spatial Intelligence",
      "type": "blog",
      "date": "2026-09-01",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Early access only."
   },
   {
    "id": "gaia-1",
    "name": "GAIA-1",
    "org": [
     "Wayve"
    ],
    "date": "2023-06",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": false,
    "summary": "A generative world model that produces driving video conditioned on video, text and vehicle actions.",
    "why_it_matters": "It was the first large generative world model for autonomous driving.",
    "url": "https://arxiv.org/abs/2309.17080",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/introducing-gaia1/",
      "title": "Introducing GAIA-1: A Cutting-Edge Generative AI Model for Autonomy",
      "type": "blog",
      "date": "2023-06-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2309.17080",
      "title": "GAIA-1: A Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2023-09-29",
      "accessed": "2026-10-10"
     }
    ],
    "note": "First announced on Wayve's blog 2023-06-17; technical report 2023-09-29. No weights released."
   },
   {
    "id": "drivedreamer",
    "name": "DriveDreamer",
    "org": [
     "GigaAI",
     "Tsinghua University"
    ],
    "date": "2023-09",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": null,
    "summary": "A driving world model trained on real driving scenes that generates controllable driving video and predicts future driving actions.",
    "why_it_matters": "It is an early Chinese driving world model; its makers later built the GigaWorld robot models.",
    "url": "https://arxiv.org/abs/2309.09777",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2309.09777",
      "title": "DriveDreamer: Towards Real-world-driven World Models for Autonomous Driving",
      "type": "paper",
      "date": "2023-09-18",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "vista",
    "name": "Vista",
    "org": [
     "HKUST",
     "OpenDriveLab (Shanghai AI Lab)",
     "University of Tübingen",
     "Tübingen AI Center",
     "University of Hong Kong"
    ],
    "date": "2024-05",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": true,
    "summary": "An open driving world model, initialised from Stable Video Diffusion, that predicts high-fidelity driving video under several kinds of action control.",
    "why_it_matters": "It is a widely used open driving world model.",
    "url": "https://arxiv.org/abs/2405.17398",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.17398",
      "title": "Vista: A Generalizable Driving World Model with High Fidelity and Versatile Controllability",
      "type": "paper",
      "date": "2024-05-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/OpenDriveLab/Vista",
      "title": "OpenDriveLab/Vista",
      "type": "repo",
      "date": "2024-06",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "gaia-2",
    "name": "GAIA-2",
    "org": [
     "Wayve"
    ],
    "date": "2025-03",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": false,
    "summary": "A multi-camera driving world model with a continuous latent space, controllable by ego actions, other agents and environment conditions.",
    "why_it_matters": "It made Wayve's world model a tool for synthetic driving data.",
    "url": "https://arxiv.org/abs/2503.20523",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.20523",
      "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2025-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://wayve.ai/thinking/gaia-2/",
      "title": "GAIA-2: Pushing the Boundaries of Video Generative Models for Safer Assisted and Automated Driving",
      "type": "blog",
      "date": "2025-03-26",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "gaia-3",
    "name": "GAIA-3",
    "org": [
     "Wayve"
    ],
    "date": "2025-12",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": false,
    "summary": "Re-drives real recorded driving scenes with controlled changes, such as a new ego trajectory, weather or vehicle, to evaluate driving models.",
    "why_it_matters": "It turned a driving world model into an evaluation tool.",
    "url": "https://wayve.ai/thinking/gaia-3/",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/gaia-3/",
      "title": "GAIA-3: Scaling World Models to Power Safety and Evaluation",
      "type": "blog",
      "date": "2025-12-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "waymo-world-model",
    "name": "Waymo World Model",
    "org": [
     "Waymo",
     "Google DeepMind"
    ],
    "date": "2026-02",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": false,
    "summary": "A driving simulator built on Genie 3 and post-trained to generate camera and lidar data, controllable by language, driving inputs and scene layouts.",
    "why_it_matters": "It is a robotaxi operator adopting a general world model for simulation.",
    "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
    "level": "verified",
    "sources": [
     {
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
      "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
      "type": "blog",
      "date": "2026-02-06",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Built by Waymo; Google DeepMind is credited for Genie 3 and acknowledged collaborators."
   },
   {
    "id": "gaia-4",
    "name": "GAIA-4",
    "org": [
     "Wayve"
    ],
    "date": "2026-08",
    "family": "action-video",
    "domains": [
     "driving"
    ],
    "open_weights": false,
    "summary": "Puts Wayve's driving model in a closed loop with the world model, so its decisions change the generated camera and radar inputs it receives next.",
    "why_it_matters": "It is the first GAIA model used for closed-loop evaluation of the full driving system.",
    "url": "https://wayve.ai/thinking/gaia-4/",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/gaia-4/",
      "title": "GAIA-4: Multimodal World Models Powering Closed-Loop Simulation for Safe and Scalable Autonomy",
      "type": "blog",
      "date": "2026-08-03",
      "accessed": "2026-10-11"
     }
    ]
   }
  ],
  "edges": [
   {
    "from": "action-conditional-atari",
    "to": "physical-interaction-video",
    "relation": "builds-on",
    "note": "The 2016 paper cites the Atari action-conditional video prediction work and moves the idea to robot pushing.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1605.07157",
      "title": "Unsupervised Learning for Physical Interaction through Video Prediction",
      "type": "paper",
      "date": "2016-05-23",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "physical-interaction-video",
    "to": "deep-visual-foresight",
    "relation": "uses",
    "note": "Deep Visual Foresight uses the same video prediction network as the 2016 paper for robot planning.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1610.00696",
      "title": "Deep Visual Foresight for Planning Robot Motion",
      "type": "paper",
      "date": "2016-10-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "action-conditional-atari",
    "to": "world-models-2018",
    "relation": "builds-on",
    "note": "World Models cites the Atari action-conditional video prediction work.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1803.10122",
      "title": "World Models",
      "type": "paper",
      "date": "2018-03-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "interaction-networks",
    "to": "gns",
    "relation": "builds-on",
    "note": "GNS cites Interaction Networks and comes from the same DeepMind group.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2002.09405",
      "title": "Learning to Simulate Complex Physics with Graph Networks",
      "type": "paper",
      "date": "2020-02-21",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gns",
    "to": "meshgraphnets",
    "relation": "builds-on",
    "note": "MeshGraphNets, from the same DeepMind group, cites GNS and compares against it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2010.03409",
      "title": "Learning Mesh-Based Simulation with Graph Networks",
      "type": "paper",
      "date": "2020-10-07",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gns",
    "to": "nerd",
    "relation": "builds-on",
    "note": "NeRD cites GNS among earlier neural simulators.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.15755",
      "title": "Neural Robot Dynamics",
      "type": "paper",
      "date": "2025-08-21",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "meshgraphnets",
    "to": "nerd",
    "relation": "builds-on",
    "note": "NeRD cites MeshGraphNets among earlier neural simulators.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.15755",
      "title": "Neural Robot Dynamics",
      "type": "paper",
      "date": "2025-08-21",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "planet",
    "relation": "builds-on",
    "note": "PlaNet reuses the image encoder and decoder networks of World Models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1811.04551",
      "title": "Learning Latent Dynamics for Planning from Pixels",
      "type": "paper",
      "date": "2018-11-12",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "dreamer",
    "relation": "builds-on",
    "note": "Dreamer reuses the encoder and decoder networks of World Models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1912.01603",
      "title": "Dream to Control: Learning Behaviors by Latent Imagination",
      "type": "paper",
      "date": "2019-12-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "planet",
    "to": "dreamer",
    "relation": "successor",
    "note": "Dreamer, by the same first author, keeps PlaNet's state-space model and learns behaviours instead of online planning.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1912.01603",
      "title": "Dream to Control: Learning Behaviors by Latent Imagination",
      "type": "paper",
      "date": "2019-12-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamer",
    "to": "dreamerv2",
    "relation": "successor",
    "note": "DreamerV2 is the next version of the Dreamer agent.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2010.02193",
      "title": "Mastering Atari with Discrete World Models",
      "type": "paper",
      "date": "2020-10-05",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamerv2",
    "to": "dreamerv3",
    "relation": "successor",
    "note": "The DreamerV3 paper describes DreamerV1 and DreamerV2 as its predecessors.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2301.04104",
      "title": "Mastering Diverse Domains through World Models",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamerv3",
    "to": "dreamer-4",
    "relation": "successor",
    "note": "Dreamer 4 cites Dreamer 3 as its predecessor and compares with it in Minecraft.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2509.24527",
      "title": "Training Agents Inside of Scalable World Models",
      "type": "paper",
      "date": "2025-09-29",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamerv2",
    "to": "daydreamer",
    "relation": "uses",
    "note": "DayDreamer runs the Dreamer algorithm (the 2019 and 2020 papers) on physical robots.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2206.14176",
      "title": "DayDreamer: World Models for Physical Robot Learning",
      "type": "paper",
      "date": "2022-06-28",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "muzero",
    "to": "td-mpc",
    "relation": "builds-on",
    "note": "TD-MPC says it learns a policy to guide planning analogous to MuZero.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2203.04955",
      "title": "Temporal Difference Learning for Model Predictive Control",
      "type": "paper",
      "date": "2022-03-09",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "td-mpc",
    "to": "td-mpc2",
    "relation": "successor",
    "note": "TD-MPC2 is a series of improvements on TD-MPC by the same authors.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2310.16828",
      "title": "TD-MPC2: Scalable, Robust World Models for Continuous Control",
      "type": "paper",
      "date": "2023-10-25",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "iris",
    "relation": "builds-on",
    "note": "IRIS calls World Models pioneering work on agents trained in imagination.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2209.00588",
      "title": "Transformers are Sample-Efficient World Models",
      "type": "paper",
      "date": "2022-09-01",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamerv2",
    "to": "iris",
    "relation": "builds-on",
    "note": "IRIS names DreamerV2 as the best Atari agent learning in imagination and targets its low sample efficiency.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2209.00588",
      "title": "Transformers are Sample-Efficient World Models",
      "type": "paper",
      "date": "2022-09-01",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "iris",
    "to": "diamond",
    "relation": "successor",
    "note": "DIAMOND comes from the same Geneva group, compares against IRIS, and replaces discrete tokens with pixel-space diffusion.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.12399",
      "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
      "type": "paper",
      "date": "2024-05-20",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "jepa-proposal",
    "to": "v-jepa",
    "relation": "builds-on",
    "note": "V-JEPA cites LeCun's 2022 JEPA proposal as the architecture it implements for video.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2404.08471",
      "title": "Revisiting Feature Prediction for Learning Visual Representations from Video",
      "type": "paper",
      "date": "2024-02-15",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "v-jepa",
    "to": "v-jepa-2",
    "relation": "successor",
    "note": "V-JEPA 2 scales the original V-JEPA recipe of Bardes et al. (2024).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09985",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "dino-wm",
    "relation": "builds-on",
    "note": "DINO-WM cites World Models when defining the dynamics models it learns.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2411.04983",
      "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
      "type": "paper",
      "date": "2024-11-07",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dino-wm",
    "to": "v-jepa-2",
    "relation": "builds-on",
    "note": "V-JEPA 2 names DINO-WM as the closest prior work and scales its planning idea to real robots.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09985",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "video-diffusion-models",
    "to": "stable-video-diffusion",
    "relation": "builds-on",
    "note": "Stable Video Diffusion cites Video Diffusion Models as prior video diffusion work.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2311.15127",
      "title": "Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets",
      "type": "paper",
      "date": "2023-11-25",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "video-diffusion-models",
    "to": "unipi",
    "relation": "builds-on",
    "note": "UniPi cites Video Diffusion Models for its video generation approach.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2302.00111",
      "title": "Learning Universal Policies via Text-Guided Video Generation",
      "type": "paper",
      "date": "2023-01-31",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "video-diffusion-models",
    "to": "unisim",
    "relation": "builds-on",
    "note": "UniSim cites Video Diffusion Models for its video generation approach.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2310.06114",
      "title": "Learning Interactive Real-World Simulators",
      "type": "paper",
      "date": "2023-10-09",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "video-diffusion-models",
    "to": "cosmos",
    "relation": "builds-on",
    "note": "The Cosmos report cites Video Diffusion Models among the video generation work it builds on.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2501.03575",
      "title": "Cosmos World Foundation Model Platform for Physical AI",
      "type": "paper",
      "date": "2025-01-07",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "unipi",
    "to": "unisim",
    "relation": "builds-on",
    "note": "UniSim cites UniPi's result that video generation can act as a policy and uses a similar inverse dynamics model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2310.06114",
      "title": "Learning Interactive Real-World Simulators",
      "type": "paper",
      "date": "2023-10-09",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "sora",
    "relation": "builds-on",
    "note": "The Sora report cites World Models among earlier generative models of video.",
    "level": "verified",
    "sources": [
     {
      "url": "https://openai.com/index/video-generation-models-as-world-simulators/",
      "title": "Video generation models as world simulators",
      "type": "blog",
      "date": "2024-02-15",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "sora",
    "to": "sora-2",
    "relation": "successor",
    "note": "OpenAI describes Sora 2 as the follow-up to the original Sora of February 2024.",
    "level": "verified",
    "sources": [
     {
      "url": "https://openai.com/index/sora-2/",
      "title": "Sora 2 is here",
      "type": "blog",
      "date": "2025-09-30",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "veo-2",
    "to": "veo-3",
    "relation": "successor",
    "note": "Google says Veo 3 improves on the quality of Veo 2.",
    "level": "verified",
    "sources": [
     {
      "url": "https://blog.google/innovation-and-ai/products/generative-media-models-io-2025/",
      "title": "Fuel your creativity with new generative media models and tools",
      "type": "blog",
      "date": "2025-05-20",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "veo-2",
    "to": "veo-robotics-sim",
    "relation": "uses",
    "note": "The Gemini Robotics evaluator is built on Veo 2.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.10675",
      "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
      "type": "paper",
      "date": "2025-12-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos",
    "to": "cosmos-predict-2-5",
    "relation": "successor",
    "note": "Cosmos-Predict2.5 is the next generation of Cosmos world foundation models and improves on Cosmos-Predict1.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2511.00062",
      "title": "World Simulation with Video Foundation Models for Physical AI",
      "type": "paper",
      "date": "2025-10-28",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos-predict-2-5",
    "to": "cosmos-3",
    "relation": "successor",
    "note": "The Cosmos 3 report compares its models with their Cosmos-Predict2.5 predecessors.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.02800",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos",
    "to": "dreamgen",
    "relation": "uses",
    "note": "DreamGen fine-tunes and benchmarks Cosmos among four video world models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.12705",
      "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
      "type": "paper",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "wan",
    "to": "dreamgen",
    "relation": "uses",
    "note": "DreamGen uses WAN2.1 as the base video world model for most robot experiments.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.12705",
      "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
      "type": "paper",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos",
    "to": "cosmos-policy",
    "relation": "uses",
    "note": "Cosmos Policy fine-tunes Cosmos-Predict2-2B, a 2025 model of the Cosmos family.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.16163",
      "title": "Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning",
      "type": "paper",
      "date": "2026-01-22",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos",
    "to": "genie-envisioner",
    "relation": "uses",
    "note": "Genie Envisioner uses Cosmos2 2B as one of its two base video models, next to LTX-Video.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.05635",
      "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
      "type": "paper",
      "date": "2025-08-07",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos",
    "to": "ge-sim-2",
    "relation": "uses",
    "note": "GE-Sim 2.0 is built on the Cosmos-Predict2-2B-Video2World model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2605.27491",
      "title": "GE-Sim 2.0: A Roadmap Towards Comprehensive Closed-loop Video World Simulators for Robotic Manipulation",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "genie-envisioner",
    "to": "ge-sim-2",
    "relation": "successor",
    "note": "GE-Sim 2.0 builds on the action-conditioned video framework of Genie Envisioner.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2605.27491",
      "title": "GE-Sim 2.0: A Roadmap Towards Comprehensive Closed-loop Video World Simulators for Robotic Manipulation",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "cosmos-predict-2-5",
    "to": "dreamdojo",
    "relation": "uses",
    "note": "DreamDojo is built on the pretrained Cosmos-Predict2.5 model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "genie",
    "to": "dreamdojo",
    "relation": "builds-on",
    "note": "DreamDojo's latent action model follows the spatiotemporal Transformer design of Genie.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "wan",
    "to": "dreamzero",
    "relation": "uses",
    "note": "DreamZero is built on a pretrained Wan image-to-video backbone.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.15922",
      "title": "World Action Models are Zero-shot Policies",
      "type": "paper",
      "date": "2026-02-17",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "dreamgen",
    "to": "dreamzero",
    "relation": "builds-on",
    "note": "DreamZero cites DreamGen as recent work showing video models can produce synthetic robot data.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.15922",
      "title": "World Action Models are Zero-shot Policies",
      "type": "paper",
      "date": "2026-02-17",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "unipi",
    "to": "1xwm-policy",
    "relation": "builds-on",
    "note": "1X says its video-then-inverse-dynamics approach follows DreamGen and UniPi.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "title": "1X World Model | From Video to Action: A New Way Robots Learn",
      "type": "blog",
      "date": "2026-01-12",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "dreamgen",
    "to": "1xwm-policy",
    "relation": "builds-on",
    "note": "1X says its approach follows DreamGen and uses a DreamGen-like two-image inverse dynamics model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "title": "1X World Model | From Video to Action: A New Way Robots Learn",
      "type": "blog",
      "date": "2026-01-12",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "sora",
    "to": "1x-world-model",
    "relation": "builds-on",
    "note": "The 1X post says it builds on advances in video generation and links to Sora.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/1x-world-model",
      "title": "1X World Model",
      "type": "blog",
      "date": "2024-09-17",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "1x-world-model",
    "to": "1xwm-policy",
    "relation": "successor",
    "note": "Both carry the name 1X World Model from the same company; the 2026 post does not refer to the 2024 model directly.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "title": "1X World Model | From Video to Action: A New Way Robots Learn",
      "type": "blog",
      "date": "2026-01-12",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "stable-video-diffusion",
    "to": "vista",
    "relation": "uses",
    "note": "Vista is initialised from Stable Video Diffusion.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.17398",
      "title": "Vista: A Generalizable Driving World Model with High Fidelity and Versatile Controllability",
      "type": "paper",
      "date": "2024-05-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "stable-video-diffusion",
    "to": "ctrl-world",
    "relation": "uses",
    "note": "Ctrl-World is initialised from the 1.5B Stable Video Diffusion checkpoint.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "unisim",
    "to": "ctrl-world",
    "relation": "builds-on",
    "note": "Ctrl-World cites UniSim among earlier work using video models for robots.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "action-conditional-atari",
    "to": "genie",
    "relation": "builds-on",
    "note": "Genie counts itself in the world-model class of Oh et al. (2015) and Ha and Schmidhuber (2018).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "date": "2024-02-23",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "genie",
    "relation": "builds-on",
    "note": "Genie counts itself in the world-model class of Ha and Schmidhuber (2018).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "date": "2024-02-23",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "unisim",
    "to": "genie",
    "relation": "builds-on",
    "note": "Genie cites UniSim as a recent large world model for robot manipulation.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "date": "2024-02-23",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "genie",
    "to": "genie-2",
    "relation": "successor",
    "note": "The Genie 2 blog presents it as the successor of Genie 1, moving from 2D to 3D worlds.",
    "level": "verified",
    "sources": [
     {
      "url": "https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/",
      "title": "Genie 2: A large-scale foundation world model",
      "type": "blog",
      "date": "2024-12-04",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "genie-2",
    "to": "genie-3",
    "relation": "successor",
    "note": "Google DeepMind says Genie 3 improves consistency and realism over Genie 2 and adds real-time interaction.",
    "level": "verified",
    "sources": [
     {
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "title": "Genie 3: A new frontier for world models",
      "type": "blog",
      "date": "2025-08-05",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "genie-3",
    "to": "waymo-world-model",
    "relation": "uses",
    "note": "Waymo says its world model is built on Genie 3.",
    "level": "verified",
    "sources": [
     {
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
      "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
      "type": "blog",
      "date": "2026-02-06",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "gamengen",
    "relation": "builds-on",
    "note": "GameNGen compares itself with the World Models DOOM simulation.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2408.14837",
      "title": "Diffusion Models Are Real-Time Game Engines",
      "type": "paper",
      "date": "2024-08-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "genie",
    "to": "gamengen",
    "relation": "builds-on",
    "note": "GameNGen cites Genie among earlier neural simulations of games.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2408.14837",
      "title": "Diffusion Models Are Real-Time Game Engines",
      "type": "paper",
      "date": "2024-08-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gamengen",
    "to": "oasis",
    "relation": "builds-on",
    "note": "The Oasis page names GameNGen as a recent action-conditioned world model and explains its different architecture choice.",
    "level": "verified",
    "sources": [
     {
      "url": "https://oasis-model.github.io/",
      "title": "Oasis: A Universe in a Transformer",
      "type": "site",
      "date": "2024-10-31",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "diamond",
    "to": "oasis",
    "relation": "builds-on",
    "note": "The Oasis page names DIAMOND as a recent action-conditioned world model and explains its different architecture choice.",
    "level": "verified",
    "sources": [
     {
      "url": "https://oasis-model.github.io/",
      "title": "Oasis: A Universe in a Transformer",
      "type": "site",
      "date": "2024-10-31",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "oasis",
    "to": "oasis-3",
    "relation": "successor",
    "note": "Decart says Oasis 3 builds on the real-time foundation of Oasis 1 and Oasis 2.",
    "level": "verified",
    "sources": [
     {
      "url": "https://decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
      "title": "Introducing Oasis 3: First Interactive World Model for Physical AI",
      "type": "blog",
      "date": "2026-06-10",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "iris",
    "to": "muse-wham",
    "relation": "builds-on",
    "note": "The WHAM paper says it builds on world-model work including Transformer models such as IRIS.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.nature.com/articles/s41586-025-08600-3",
      "title": "World and Human Action Models towards gameplay ideation (Nature)",
      "type": "paper",
      "date": "2025-02-19",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "muse-wham",
    "relation": "builds-on",
    "note": "The WHAM paper says it builds on the line of work on world models started by Ha and Schmidhuber.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.nature.com/articles/s41586-025-08600-3",
      "title": "World and Human Action Models towards gameplay ideation (Nature)",
      "type": "paper",
      "date": "2025-02-19",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "oasis",
    "to": "matrix-game-2",
    "relation": "builds-on",
    "note": "Matrix-Game 2.0 cites Oasis as earlier real-time autoregressive work that lacked full interactivity.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.13009",
      "title": "Matrix-game 2.0: An open-source, real-time, and streaming interactive world model",
      "type": "paper",
      "date": "2025-08-18",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "wan",
    "to": "matrix-game-2",
    "relation": "builds-on",
    "note": "Matrix-Game 2.0 starts from SkyReels-V2, which follows the Wan 2.1 architecture.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2508.13009",
      "title": "Matrix-game 2.0: An open-source, real-time, and streaming interactive world model",
      "type": "paper",
      "date": "2025-08-18",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "matrix-game-2",
    "to": "matrix-game-3",
    "relation": "successor",
    "note": "Matrix-Game 3.0 is built on Matrix-Game 2.0.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.08995",
      "title": "Matrix-Game 3.0: Real-Time and Streaming Interactive World Model with Long-Horizon Memory",
      "type": "paper",
      "date": "2026-04-10",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "wan",
    "to": "lingbot-world",
    "relation": "uses",
    "note": "LingBot-World adopts the 14B Wan2.2 image-to-video model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.20540",
      "title": "Advancing Open-source World Models",
      "type": "paper",
      "date": "2026-01-28",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "hunyuanworld-1",
    "to": "hy-world-1-5",
    "relation": "successor",
    "note": "Tencent's report says HY-World 1.5 adds real-time interaction that HunyuanWorld 1.0 lacked.",
    "level": "verified",
    "sources": [
     {
      "url": "https://3d-models.hunyuan.tencent.com/world/world1_5/HYWorld_1.5_Tech_Report.pdf",
      "title": "HY-World 1.5 Technical Report",
      "type": "paper",
      "date": "2025-12-17",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "odyssey-1",
    "to": "odyssey-3",
    "relation": "successor",
    "note": "Same company and numbered series (Odyssey-2 came in October 2025); the Odyssey-3 post does not name earlier models.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://odyssey.systems/introducing-odyssey-3",
      "title": "Introducing Odyssey-3: A General-Purpose Physical Intelligence",
      "type": "blog",
      "date": "2026-09-15",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "marble",
    "to": "atlas",
    "relation": "successor",
    "note": "World Labs calls Atlas its next-generation world model that will power future versions of Marble.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.worldlabs.ai/blog/atlas",
      "title": "Atlas: A World Model for Spatial Intelligence",
      "type": "blog",
      "date": "2026-09-01",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "world-models-2018",
    "to": "gaia-1",
    "relation": "builds-on",
    "note": "GAIA-1 cites Ha and Schmidhuber when introducing world models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2309.17080",
      "title": "GAIA-1: A Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2023-09-29",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gaia-1",
    "to": "gaia-2",
    "relation": "successor",
    "note": "GAIA-2 describes GAIA-1 as its predecessor.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.20523",
      "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2025-03-26",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gaia-2",
    "to": "gaia-3",
    "relation": "successor",
    "note": "Wayve's GAIA-3 post describes GAIA-1 and GAIA-2 as the earlier steps.",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/gaia-3/",
      "title": "GAIA-3: Scaling World Models to Power Safety and Evaluation",
      "type": "blog",
      "date": "2025-12-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "gaia-3",
    "to": "gaia-4",
    "relation": "successor",
    "note": "Wayve calls GAIA-4 the latest step in its GAIA line after GAIA-1, 2 and 3.",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/gaia-4/",
      "title": "GAIA-4: Multimodal World Models Powering Closed-Loop Simulation for Safe and Scalable Autonomy",
      "type": "blog",
      "date": "2026-08-03",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "from": "dreamerv3",
    "to": "drivedreamer",
    "relation": "builds-on",
    "note": "DriveDreamer cites World Models and the Dreamer series as the world-model work behind it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2309.09777",
      "title": "DriveDreamer: Towards Real-world-driven World Models for Autonomous Driving",
      "type": "paper",
      "date": "2023-09-18",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "gaia-1",
    "to": "vista",
    "relation": "builds-on",
    "note": "Vista cites GAIA-1 as an earlier driving world model and compares against it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.17398",
      "title": "Vista: A Generalizable Driving World Model with High Fidelity and Versatile Controllability",
      "type": "paper",
      "date": "2024-05-27",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "from": "drivedreamer",
    "to": "vista",
    "relation": "builds-on",
    "note": "Vista cites DriveDreamer as an earlier driving world model and compares against it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2405.17398",
      "title": "Vista: A Generalizable Driving World Model with High Fidelity and Versatile Controllability",
      "type": "paper",
      "date": "2024-05-27",
      "accessed": "2026-10-10"
     }
    ]
   }
  ],
  "eras": [
   {
    "from": "2015",
    "to": "2017",
    "name": "Predicting what happens next",
    "summary": "Researchers trained networks to predict game frames or robot camera images from actions, and to predict object physics with graph networks."
   },
   {
    "from": "2018",
    "to": "2022",
    "name": "Learning inside a world model",
    "summary": "Agents such as World Models, PlaNet, Dreamer, MuZero and TD-MPC learned compact models of their environment and planned or trained inside them."
   },
   {
    "from": "2023",
    "to": "2024",
    "name": "Video models as simulators",
    "summary": "Large video generators such as UniSim, GAIA-1, Sora and Genie were trained on internet or driving video and presented as simulators of the world."
   },
   {
    "from": "2025",
    "to": "2025",
    "name": "Foundation and real-time world models",
    "summary": "Companies released open world foundation models (Cosmos, Wan) and real-time interactive worlds (Genie 3, Matrix-Game 2.0), and robot teams used them for data and evaluation."
   },
   {
    "from": "2026",
    "to": "2026",
    "name": "World models as robot policies and test tracks",
    "summary": "World models were used directly as robot policies (DreamZero, Cosmos Policy, 1XWM) and as closed-loop simulators for robots and cars (GAIA-4, Oasis 3, Waymo World Model)."
   }
  ]
 },
 "methods": {
  "families": [
   {
    "id": "latent-dynamics",
    "name": "Latent dynamics models",
    "plain": "A latent dynamics model compresses each observation, such as a camera image, into a short list of numbers and learns to predict how that list changes after each action. It takes past observations and actions as input and outputs the predicted next state, usually together with the reward the agent will receive. It is used to plan actions or to train a decision-making program (a policy) on predicted experience, so the agent needs fewer real trials.",
    "how_it_works": [
     "The agent acts in a game, a simulator or the real world and records what it observed, which action it took and what reward it received.",
     "An encoder network compresses each observation into a short list of numbers called the latent state.",
     "A dynamics network learns to predict the next latent state and the reward from the current latent state and the chosen action.",
     "The agent plans, or trains its policy, by running many predicted futures inside the model instead of in the real environment.",
     "The agent then acts with the improved policy, records new data and updates the model, and the cycle repeats."
    ],
    "inputs": [
     "observations such as camera images or joint readings",
     "actions",
     "rewards (a score the task gives for good outcomes)"
    ],
    "outputs": [
     "predicted next latent state",
     "predicted reward",
     "in some models also a predicted image, a value estimate or a recommended action"
    ],
    "actions": "yes",
    "strengths": [
     "Predicting a short list of numbers is cheaper than predicting full images, so a planner can test many candidate actions quickly. The 2026 robotic manipulation survey (arXiv 2606.00113) gives this as the main appeal of latent models.",
     "Learning needs fewer real trials because most practice happens inside the model. DayDreamer reports a four-legged robot learning to roll off its back, stand up and walk from scratch in 1 hour, without resets.",
     "One method can cover many tasks with one setting. DreamerV3 reports results on over 150 diverse tasks with a single configuration, and TD-MPC2 reports a single 317M parameter agent performing 80 tasks.",
     "The approach works without being told the rules. MuZero learned Go, chess, shogi and 57 Atari games without knowledge of their underlying dynamics."
    ],
    "limits": [
     "Errors add up when the model predicts many steps ahead, because each step starts from the previous prediction. The 2025 embodied AI survey (arXiv 2510.16732) names this error accumulation as the main weakness of step-by-step prediction.",
     "Compressing an image into a short list of numbers can drop small visual details that matter for the task. The DIAMOND authors make this point about compact discrete latent states.",
     "People cannot easily read the internal numbers, so it is hard to check whether a prediction is physically plausible. The 2026 robotic manipulation survey calls this a loss of auditability.",
     "A model trained for one task or one robot often transfers poorly to others. The 2025 embodied AI survey and the 2026 robotic manipulation survey both report this for task-specific world models."
    ],
    "inferred_points": [],
    "examples": [
     {
      "name": "World Models",
      "org": "Google Brain; NNAISENSE; Swiss AI Lab IDSIA (USI & SUPSI)",
      "year": 2018,
      "date": "2018-03-27",
      "url": "https://arxiv.org/abs/1803.10122",
      "level": "verified",
      "note": "Trained an agent entirely inside environments generated by its world model and transferred the policy back to the actual environment."
     },
     {
      "name": "PlaNet",
      "org": "Google Brain; DeepMind; Google Research; University of Toronto; University of Michigan",
      "year": 2018,
      "date": "2018-11-12",
      "url": "https://arxiv.org/abs/1811.04551",
      "level": "verified",
      "note": "Learns dynamics from images and chooses actions by planning in latent space."
     },
     {
      "name": "MuZero",
      "org": "DeepMind",
      "year": 2019,
      "date": "2019-11-19",
      "url": "https://arxiv.org/abs/1911.08265",
      "level": "verified",
      "note": "Predicts only reward, action-selection policy and value; it does not reconstruct images."
     },
     {
      "name": "DayDreamer",
      "org": "University of California, Berkeley",
      "year": 2022,
      "date": "2022-06-28",
      "url": "https://arxiv.org/abs/2206.14176",
      "level": "verified",
      "note": "Applies Dreamer to 4 physical robots learning directly in the real world."
     },
     {
      "name": "DreamerV3",
      "org": "Google DeepMind; University of Toronto",
      "year": 2023,
      "date": "2023-01-10",
      "url": "https://arxiv.org/abs/2301.04104",
      "level": "verified"
     },
     {
      "name": "TD-MPC2",
      "org": "University of California San Diego",
      "year": 2023,
      "date": "2023-10-25",
      "url": "https://arxiv.org/abs/2310.16828",
      "level": "verified",
      "note": "Plans in the latent space of a decoder-free world model."
     }
    ],
    "used_for": [
     "planning",
     "training policies",
     "robot control",
     "games"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/1803.10122",
      "title": "World Models",
      "type": "paper",
      "date": "2018-03-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/1811.04551",
      "title": "Learning Latent Dynamics for Planning from Pixels",
      "type": "paper",
      "date": "2018-11-12",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/1911.08265",
      "title": "Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model",
      "type": "paper",
      "date": "2019-11-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2206.14176",
      "title": "DayDreamer: World Models for Physical Robot Learning",
      "type": "paper",
      "date": "2022-06-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2301.04104",
      "title": "Mastering Diverse Domains through World Models",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2310.16828",
      "title": "TD-MPC2: Scalable, Robust World Models for Continuous Control",
      "type": "paper",
      "date": "2023-10-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2405.12399",
      "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
      "type": "paper",
      "date": "2024-05-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.16732v3",
      "title": "A Comprehensive Survey on World Models for Embodied AI",
      "type": "paper",
      "date": "2025-10-19",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section III)."
     },
     {
      "url": "https://arxiv.org/html/2606.00113v1",
      "title": "World Models for Robotic Manipulation: A Survey",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section III-B)."
     }
    ],
    "level": "inferred"
   },
   {
    "id": "video-generation",
    "name": "Video generators",
    "plain": "A video generator creates a new video clip from a text description, a starting image or a short video, and it does not receive step-by-step actions from a player or a robot. Its output is the video clip. It is used to make video content and as a base model that builders later fine-tune into action-conditioned world models, and OpenAI describes scaling such models as a path towards general-purpose simulators of the physical world.",
    "how_it_works": [
     "Builders collect very large sets of videos and images and attach a text description to each. NVIDIA reports about 20M hours of raw video for Cosmos, and OpenAI trained a captioning model to describe every video in Sora's training set.",
     "A compression network shrinks each video into a smaller grid of numbers so the generator can process long, high-resolution clips.",
     "The generator learns to turn random noise into a clip that matches the description, or to continue a given clip.",
     "At use time a person supplies a prompt, a start image or a clip, and the model produces the new clip.",
     "Builders can fine-tune the same model on robot or driving video with camera paths or actions as extra inputs, which turns it into an action-conditioned model. The Cosmos paper shows such fine-tuning for camera control, robotic manipulation and autonomous driving."
    ],
    "inputs": [
     "text prompt",
     "starting image",
     "starting video clip"
    ],
    "outputs": [
     "new video clip",
     "sometimes still images"
    ],
    "actions": "no",
    "strengths": [
     "Training needs only videos, images and text descriptions, which exist in very large amounts. NVIDIA built Cosmos from about 20M hours of raw video.",
     "Large models show some 3D consistency and object permanence without being designed for it. OpenAI reports that Sora keeps people and scene elements consistent as the camera moves, and often, though not always, keeps objects that are hidden or leave the frame.",
     "One model can perform visual tasks it was not trained for. Google DeepMind reports Veo 3 segmenting objects, detecting edges, editing images and solving mazes without task-specific training.",
     "A general model can be specialised for a robot or a car with less data. The Cosmos paper states that the dataset for post-training can be much smaller than the pretraining data.",
     "Generated videos can become robot training data. NVIDIA's DreamGen adapts image-to-video models to a robot, recovers actions from the generated videos, and reports a humanoid robot performing 22 new behaviors while its teleoperation data covered a single pick-and-place task in one environment."
    ],
    "limits": [
     "Generated clips can break physics. OpenAI states that Sora does not accurately model the physics of many basic interactions, such as glass shattering.",
     "Long clips can drift. OpenAI lists incoherence in long samples and objects that appear spontaneously, and NVIDIA reports objects unexpectedly appearing from below in some outputs of its autoregressive Cosmos models.",
     "The model has no input for a specific robot action, so it cannot directly show what happens if a robot moves its arm a certain way.",
     "Uses beyond video content are often proposals. The Cosmos paper lists policy evaluation, policy training, planning and synthetic data generation as uses, and states that it does not include empirical results for them.",
     "Builders often do not publish the model details. OpenAI's Sora report states that model and implementation details are not included."
    ],
    "inferred_points": [
     "The model has no input for a specific robot action, so it cannot directly show what happens if a robot moves its arm a certain way. (Inferred from the inputs that OpenAI and NVIDIA list, which are text, images and video.)"
    ],
    "examples": [
     {
      "name": "Sora",
      "org": "OpenAI",
      "year": 2024,
      "date": "2024-02-15",
      "url": "https://web.archive.org/web/20240216004954id_/https://openai.com/research/video-generation-models-as-world-simulators",
      "level": "verified",
      "note": "Technical report titled Video generation models as world simulators. openai.com returned HTTP 403 to automated requests; the text was read from the Internet Archive copy captured 2024-02-16."
     },
     {
      "name": "Cosmos world foundation models (Cosmos-Predict1)",
      "org": "NVIDIA",
      "year": 2025,
      "date": "2025-01-07",
      "url": "https://arxiv.org/abs/2501.03575",
      "level": "verified",
      "note": "Includes diffusion and autoregressive models (Text2World, Video2World) and fine-tuned versions for camera control, robotic manipulation and autonomous driving."
     },
     {
      "name": "DreamGen",
      "org": "NVIDIA, with University of Washington, KAIST, UCLA, UCSD, CalTech, NTU, University of Maryland and UT Austin",
      "year": 2025,
      "date": "2025-05-19",
      "url": "https://arxiv.org/abs/2505.12705",
      "level": "verified",
      "note": "Uses image-to-video models adapted to a robot to generate training videos, then labels them with actions using a separate model."
     },
     {
      "name": "Veo 3 (studied in Video models are zero-shot learners and reasoners)",
      "org": "Google DeepMind",
      "year": 2025,
      "date": "2025-09-24",
      "url": "https://arxiv.org/abs/2509.20328",
      "level": "verified",
      "note": "The paper tests what Google DeepMind's Veo 3 video model can do without task-specific training."
     }
    ],
    "used_for": [
     "content creation",
     "base model for other world models",
     "generating training data"
    ],
    "sources": [
     {
      "url": "https://web.archive.org/web/20240216004954id_/https://openai.com/research/video-generation-models-as-world-simulators",
      "title": "Video generation models as world simulators",
      "type": "blog",
      "date": "2024-02-15",
      "accessed": "2026-10-10",
      "note": "OpenAI technical report read from the Internet Archive copy because openai.com returned HTTP 403."
     },
     {
      "url": "https://arxiv.org/abs/2501.03575",
      "title": "Cosmos World Foundation Model Platform for Physical AI",
      "type": "paper",
      "date": "2025-01-07",
      "accessed": "2026-10-10",
      "note": "Sections 2.1, 3.1, 5.2.7 and 6 read in the HTML full text."
     },
     {
      "url": "https://arxiv.org/abs/2505.12705",
      "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
      "type": "paper",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2509.20328",
      "title": "Video models are zero-shot learners and reasoners",
      "type": "paper",
      "date": "2025-09-24",
      "accessed": "2026-10-10"
     }
    ],
    "level": "inferred"
   },
   {
    "id": "action-video",
    "name": "Action-conditioned video models",
    "plain": "An action-conditioned video model predicts the next video frames of a scene given the actions that a robot, a car or a game agent will take. It takes recent camera frames and a sequence of planned actions as input and outputs the video those actions would produce. Teams use it to test robot and driving software without the real machine, to create training data and to train policies inside the predictions.",
    "how_it_works": [
     "Collect videos from robots, cars or games together with the action taken at each moment, such as arm positions, steering or game controls.",
     "Start from a video model, often one pretrained on general video, and train it to predict the next frames from past frames plus the recorded actions.",
     "To test a policy, feed the policy's actions into the model, show the policy the predicted frames, and repeat step by step.",
     "Judge whether the predicted video shows the task succeeding, and compare policies by their predicted success.",
     "Successful predicted runs can also be used as extra training examples for the policy."
    ],
    "inputs": [
     "recent camera frames from one or more cameras",
     "planned actions such as robot arm poses, steering and speed, or game controls",
     "sometimes a text instruction or a scene description"
    ],
    "outputs": [
     "predicted future video frames, sometimes for several cameras at once"
    ],
    "actions": "yes",
    "strengths": [
     "Policies can be compared without running real robots. The Ctrl-World authors report that their model can accurately rank policy performance without real-world robot rollouts, and Google DeepMind checked its Veo-based evaluator against 1600+ real-world evaluations of eight Gemini Robotics policy checkpoints and five tasks.",
     "Rare or unsafe situations can be produced on request. Wayve built GAIA-2 to simulate both common and rare driving scenarios, and the Veo-based evaluator edits scenes to add new objects, backgrounds and distractors.",
     "Predicted runs can improve a policy. Ctrl-World reports that fine-tuning on successful trajectories generated in the model improved policy success by 44.7%.",
     "Policies trained inside the model can work on real robots. The UniSim authors report vision-language and reinforcement learning policies that were deployed in the real world zero-shot after training purely in their simulator."
    ],
    "limits": [
     "The model can show events that cannot happen. The UniSim authors report hallucinations when an action does not fit the scene, and 1X reports many generations that fail to adhere to physical laws.",
     "Objects can change shape or colour or disappear during interaction, as 1X reports for its world model.",
     "Memory is short. The UniSim authors give the example of an apple that disappears from a drawer when the frames showing it being put there are no longer in the model's input.",
     "Accuracy drops for robots, scenes and tasks that are missing from the training data, which the UniSim authors list as limited out-of-domain generalization.",
     "A video that looks right can still show a physically wrong result for the robot. The 2026 robotic manipulation survey states that visual plausibility is not equivalent to action validity."
    ],
    "inferred_points": [],
    "examples": [
     {
      "name": "GAIA-1",
      "org": "Wayve",
      "year": 2023,
      "date": "2023-09-29",
      "url": "https://arxiv.org/abs/2309.17080",
      "level": "verified",
      "note": "Driving world model that takes video, text and action inputs."
     },
     {
      "name": "UniSim (Learning Interactive Real-World Simulators)",
      "org": "UC Berkeley; Google DeepMind; MIT; University of Alberta",
      "year": 2023,
      "date": "2023-10-09",
      "url": "https://arxiv.org/abs/2310.06114",
      "level": "verified",
      "note": "Name collision with Waabi's UniSim (CVPR 2023), which is listed under 3d-world."
     },
     {
      "name": "1X World Model",
      "org": "1X Technologies",
      "year": 2024,
      "date": "2024-09-17",
      "url": "https://www.1x.tech/discover/1x-world-model",
      "level": "verified",
      "note": "Predicts video from starting frames and a proposed robot action trajectory; built for evaluating policies. 1X also released a dataset and launched the 1X World Model Challenge."
     },
     {
      "name": "GAIA-2",
      "org": "Wayve",
      "year": 2025,
      "date": "2025-03-26",
      "url": "https://arxiv.org/abs/2503.20523",
      "level": "verified",
      "note": "Multi-camera driving video conditioned on ego-vehicle dynamics, other agents, environment and road layout."
     },
     {
      "name": "Ctrl-World",
      "org": "Stanford University; Tsinghua University",
      "year": 2025,
      "date": "2025-10-11",
      "url": "https://arxiv.org/abs/2510.10125",
      "level": "verified",
      "note": "Trained on the DROID dataset (95k trajectories, 564 scenes)."
     },
     {
      "name": "Veo world simulator for Gemini Robotics policy evaluation",
      "org": "Google DeepMind (Gemini Robotics Team)",
      "year": 2025,
      "date": "2025-12-11",
      "url": "https://arxiv.org/abs/2512.10675",
      "level": "verified",
      "note": "Built on the Veo video model, with robot action conditioning and multi-view consistency."
     }
    ],
    "used_for": [
     "evaluating policies",
     "training policies",
     "generating training data",
     "testing self-driving software"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/2309.17080",
      "title": "GAIA-1: A Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2023-09-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2310.06114",
      "title": "Learning Interactive Real-World Simulators",
      "type": "paper",
      "date": "2023-10-09",
      "accessed": "2026-10-10",
      "note": "Section 6 (Limitations) read in the HTML full text."
     },
     {
      "url": "https://www.1x.tech/discover/1x-world-model",
      "title": "1X World Model",
      "type": "blog",
      "date": "2024-09-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2503.20523",
      "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
      "type": "paper",
      "date": "2025-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2512.10675",
      "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
      "type": "paper",
      "date": "2025-12-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.00113v1",
      "title": "World Models for Robotic Manipulation: A Survey",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section III-A)."
     }
    ],
    "level": "inferred"
   },
   {
    "id": "interactive-world",
    "name": "Real-time interactive worlds",
    "plain": "A real-time interactive world model draws a game-like world one frame at a time while a person or an AI agent controls it with a keyboard, a mouse or text commands. It takes the latest control input and the frames it has already drawn, and it outputs the next frame fast enough for live play. It is used in games and interactive media research and as a place to train and test AI agents.",
    "how_it_works": [
     "Collect many hours of gameplay or video. GameNGen recorded an AI agent playing DOOM, Matrix-Game 2.0 produced about 1200 hours of video from Unreal Engine and GTA5 environments, and Genie learned from unlabelled Internet videos.",
     "Train a model to predict the next frame from recent frames and the control pressed at that moment. Genie learns its own small set of controls from videos that have no control labels.",
     "Make each frame fast enough for live play, for example by reducing the number of generation steps per frame.",
     "During play, each new frame joins the model's recent history, the player reacts, and the loop repeats."
    ],
    "inputs": [
     "keyboard, mouse or controller input at each frame",
     "a text prompt or image that sets up the world",
     "the frames generated so far",
     "in Genie 3, text commands that change the world"
    ],
    "outputs": [
     "the next video frame, shown live"
    ],
    "actions": "yes",
    "strengths": [
     "Worlds can be started from a prompt without hand-built 3D assets. Genie can be prompted with text, synthetic images, photographs and sketches, and Genie 3 generates worlds from a text prompt.",
     "Live play is possible on current hardware. GameNGen runs at 20 frames per second on a single TPU, Oasis at 20 frames per second, Matrix-Game 2.0 at 25 FPS and Genie 3 at 24 frames per second at 720p.",
     "Short clips can be hard to tell apart from the real game. The GameNGen authors report that human raters are only slightly better than random chance at distinguishing short clips of the game from clips of the simulation.",
     "The worlds can host AI agents. Google DeepMind ran its SIMA agent inside Genie 3 worlds and gave it goals to pursue."
    ],
    "limits": [
     "Memory is short. GameNGen has access to a little over 3 seconds of history, the first Genie model is limited to 16 frames of memory, and Oasis lists limited memory over long horizons.",
     "Sessions last minutes. Google DeepMind states that Genie 3 supports a few minutes of continuous interaction.",
     "Control is limited. Genie 3 lists a limited action space and difficulty simulating other agents, and Oasis lists difficulty with precise inventory control.",
     "Speed is hard to reach. The first Genie model ran at around 1FPS, and the Matrix-Game 2.0 authors say earlier interactive models were held back by lengthy inference steps.",
     "Access can be restricted. Genie 3 was announced as a limited research preview, and the Genie authors chose not to release model checkpoints or training data."
    ],
    "inferred_points": [],
    "examples": [
     {
      "name": "Genie",
      "org": "Google DeepMind (one author also at University of British Columbia)",
      "year": 2024,
      "date": "2024-02-23",
      "url": "https://arxiv.org/abs/2402.15391",
      "level": "verified",
      "note": "11B parameters. Runs at around 1FPS according to its authors, so it is below playable speed; listed here as the first model of the Genie line."
     },
     {
      "name": "DIAMOND",
      "org": "University of Geneva; University of Edinburgh; Microsoft Research",
      "year": 2024,
      "date": "2024-05-20",
      "url": "https://arxiv.org/abs/2405.12399",
      "level": "verified",
      "note": "Mainly a method for training game-playing agents inside a diffusion world model (mean human normalized score 1.46 on Atari 100k). Its Counter-Strike: Global Offensive model, trained on 95 hours of human gameplay, runs at 10Hz on an RTX 3090."
     },
     {
      "name": "GameNGen",
      "org": "Google Research; Google DeepMind; Tel Aviv University",
      "year": 2024,
      "date": "2024-08-27",
      "url": "https://arxiv.org/abs/2408.14837",
      "level": "verified",
      "note": "Simulates the game DOOM."
     },
     {
      "name": "Oasis",
      "org": "Decart; Etched",
      "year": 2024,
      "date": "2024-10-31",
      "url": "https://oasis-model.github.io/",
      "level": "verified",
      "note": "Takes keyboard input; the page describes building, breaking blocks and inventory without naming the game it was trained on."
     },
     {
      "name": "Genie 3",
      "org": "Google DeepMind",
      "year": 2025,
      "date": "2025-08-05",
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "level": "verified"
     },
     {
      "name": "Matrix-Game 2.0",
      "org": "Skywork AI",
      "year": 2025,
      "date": "2025-08-18",
      "url": "https://arxiv.org/abs/2508.13009",
      "level": "verified",
      "note": "Model weights and code released by the authors."
     }
    ],
    "used_for": [
     "games",
     "content creation",
     "training policies",
     "evaluating policies"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "date": "2024-02-23",
      "accessed": "2026-10-10",
      "note": "Conclusion and Broader Impact read in the HTML full text."
     },
     {
      "url": "https://arxiv.org/abs/2405.12399",
      "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
      "type": "paper",
      "date": "2024-05-20",
      "accessed": "2026-10-10",
      "note": "Section 6 read in the HTML full text."
     },
     {
      "url": "https://arxiv.org/abs/2408.14837",
      "title": "Diffusion Models Are Real-Time Game Engines",
      "type": "paper",
      "date": "2024-08-27",
      "accessed": "2026-10-10",
      "note": "Section 7 (Limitations) read in the HTML full text."
     },
     {
      "url": "https://oasis-model.github.io/",
      "title": "Oasis: A Universe in a Transformer",
      "type": "site",
      "date": "2024-10-31",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "title": "Genie 3: A new frontier for world models",
      "type": "blog",
      "date": "2025-08-05",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2508.13009",
      "title": "Matrix-game 2.0: An open-source, real-time, and streaming interactive world model",
      "type": "paper",
      "date": "2025-08-18",
      "accessed": "2026-10-10"
     }
    ],
    "level": "inferred"
   },
   {
    "id": "predictive-representation",
    "name": "Predictive representation models",
    "plain": "A predictive representation model hides part of an image or video and learns to predict a numeric description (features) of the hidden or future part, without drawing any pixels. It takes images or video, sometimes with robot actions, and outputs predicted features. It is used as a general model of what video shows, and with actions added it is used to plan robot movements toward a goal image.",
    "how_it_works": [
     "An encoder network turns images or video frames into features.",
     "Parts of the input are hidden, or lie in the future, and a predictor network estimates the features of those parts from the visible parts.",
     "Training compares the estimate with the encoder's features of the real hidden parts and adjusts both networks. Extra rules stop the networks from giving every input the same features, a failure called collapse.",
     "For robot control, an action-conditioned predictor is trained on robot video. V-JEPA 2-AC used less than 62 hours of unlabeled robot videos from the Droid dataset.",
     "To plan, the system tries candidate action sequences, predicts the resulting features and picks the sequence whose prediction is closest to the features of a goal image."
    ],
    "inputs": [
     "images or video",
     "robot actions (optional)",
     "a goal image when used for planning"
    ],
    "outputs": [
     "predicted features (numbers that describe the scene)",
     "a chosen action sequence when used for planning"
    ],
    "actions": "optional",
    "strengths": [
     "No computing is spent on drawing texture and lighting. Meta's V-JEPA 2 and NYU's DINO-WM both predict features without reconstructing pixels.",
     "Most training uses video without action labels. V-JEPA 2 was pretrained on over 1 million hours of internet video and then needed less than 62 hours of robot video for its action-conditioned part.",
     "Robots can be controlled in new places without new data. Meta deployed V-JEPA 2-AC zero-shot on Franka arms in two different labs for picking and placing objects, without collecting data from the robots in these environments.",
     "Small models can plan quickly. The LeWorldModel authors report about 15M parameters, training on a single GPU in a few hours, and planning up to 48x faster than foundation-model-based world models."
    ],
    "limits": [
     "The predictions are lists of numbers, so people cannot watch them to check whether they are right.",
     "Long plans are hard. The V-JEPA 2 authors report that prediction accuracy falls in longer rollouts and that the number of possible action sequences grows exponentially with the planning horizon.",
     "Results depend on the physical setup. V-JEPA 2-AC was sensitive to camera position, and its authors tried camera positions by hand before finding one that worked.",
     "Goals are usually given as images. The V-JEPA 2 authors note that language would be a more natural way to state goals for robots in everyday settings.",
     "Training can fail through collapse. The LeWorldModel authors describe earlier methods as fragile and dependent on extra losses, moving-average tricks, pretrained encoders or extra supervision to avoid it."
    ],
    "inferred_points": [
     "The predictions are lists of numbers, so people cannot watch them to check whether they are right. (Inferred because V-JEPA 2 and DINO-WM predict features and do not reconstruct pixels.)"
    ],
    "examples": [
     {
      "name": "I-JEPA",
      "org": "Meta AI (FAIR), with McGill University, Mila and New York University",
      "year": 2023,
      "date": "2023-01-19",
      "url": "https://arxiv.org/abs/2301.08243",
      "level": "verified",
      "note": "Image version of the joint-embedding predictive architecture (JEPA); predicts features of hidden image blocks from one visible block."
     },
     {
      "name": "DINO-WM",
      "org": "New York University (Courant Institute); Meta AI",
      "year": 2024,
      "date": "2024-11-07",
      "url": "https://arxiv.org/abs/2411.04983",
      "level": "verified",
      "note": "Predicts future features from a pretrained encoder (DINOv2) given actions, and plans toward goal features. It sits on the boundary with latent-dynamics."
     },
     {
      "name": "V-JEPA 2 and V-JEPA 2-AC",
      "org": "FAIR at Meta, with Mila and Polytechnique Montréal",
      "year": 2025,
      "date": "2025-06-11",
      "url": "https://arxiv.org/abs/2506.09985",
      "level": "verified"
     },
     {
      "name": "LeWorldModel",
      "org": "Mila & Université de Montréal; New York University; Samsung SAIL; Brown University",
      "year": 2026,
      "date": "2026-03-13",
      "url": "https://arxiv.org/abs/2603.19312",
      "level": "verified",
      "note": "A JEPA trained end to end from pixels with actions. The 2025 embodied AI survey (v3) lists it with compact-latent, task-coupled models, so it also sits on the boundary with latent-dynamics."
     }
    ],
    "used_for": [
     "pretraining visual models",
     "planning",
     "robot control"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/2301.08243",
      "title": "Self-Supervised Learning from Images with a Joint-Embedding Predictive Architecture",
      "type": "paper",
      "date": "2023-01-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2411.04983",
      "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
      "type": "paper",
      "date": "2024-11-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2506.09985",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10",
      "note": "Section 4.3 (Limitations) read in the HTML full text."
     },
     {
      "url": "https://arxiv.org/abs/2603.19312",
      "title": "LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels",
      "type": "paper",
      "date": "2026-03-13",
      "accessed": "2026-10-10"
     }
    ],
    "level": "inferred"
   },
   {
    "id": "3d-world",
    "name": "3D and 4D world models",
    "plain": "A 3D world model builds a scene as 3D structure, such as points, surfaces or a grid of occupied space, so the scene stays the same when seen from a new angle. It takes text, images, video or recorded sensor data and outputs a 3D scene that can be explored, rendered from any viewpoint or exported to games and simulators. Some versions also predict how the 3D scene will change over the next seconds, which self-driving research uses to forecast the road ahead.",
    "how_it_works": [
     "Take a prompt or a real recording. Marble accepts text, images, video or coarse 3D layouts, WonderWorld starts from a single image, and Waabi's UniSim starts from one recorded drive with camera and LiDAR data.",
     "Estimate depth and shape. Some systems first generate a 360-degree panorama, and HunyuanWorld 1.0 uses panoramic images as stand-ins for the whole world.",
     "Store the result in a 3D format such as Gaussian splats (a large set of small semi-transparent particles), meshes, or grids that mark which spaces are occupied.",
     "Render new views, let a user move through the scene, export it to a game engine, or edit it, for example by adding or removing cars.",
     "In forecasting versions such as OccWorld, a second model predicts the next occupancy grid and the car's own path from the previous grids."
    ],
    "inputs": [
     "text",
     "images",
     "video",
     "recorded sensor logs (camera and LiDAR)",
     "coarse 3D layouts"
    ],
    "outputs": [
     "3D scene as Gaussian splats, meshes or occupancy grids",
     "images or video rendered from new viewpoints",
     "predicted future occupancy grids and vehicle path (forecasting versions)"
    ],
    "actions": "optional",
    "strengths": [
     "The scene is consistent from every angle because it is stored in 3D. The HunyuanWorld 1.0 authors give geometric consistency as the advantage of 3D-based methods over video-based ones.",
     "Results move into existing tools. HunyuanWorld 1.0 exports meshes, and Marble exports Gaussian splats, meshes and videos.",
     "A recorded drive can be replayed with changes. Waabi's UniSim converts a single recorded log into a closed-loop multi-sensor simulation in which actors can be added, removed or moved.",
     "Generation can be fast enough for interactive editing. WonderWorld generates connected 3D scenes in less than 10 seconds on a single A6000 GPU."
    ],
    "limits": [
     "There is less 3D training data than video. The HunyuanWorld 1.0 authors say 3D-based methods struggle with limited training data and memory-inefficient representations.",
     "Fast movement is hard to represent. The 2025 embodied AI survey reports that scenes built from renderable pieces such as Gaussian splats have limited ability to handle rapid dynamics or changes in shape.",
     "Physics support is basic in current products. World Labs describes Marble's collider meshes as low-fidelity meshes intended for coarse physics simulation.",
     "Text and single-image prompts give limited control over details, as World Labs states for Marble.",
     "Accurate 3D needs depth sensors or many camera views, which the 2026 robotic manipulation survey notes are less available than ordinary video."
    ],
    "inferred_points": [],
    "examples": [
     {
      "name": "UniSim (A Neural Closed-Loop Sensor Simulator)",
      "org": "Waabi; University of Toronto; MIT",
      "year": 2023,
      "date": "2023-06",
      "url": "https://openaccess.thecvf.com/content/CVPR2023/html/Yang_UniSim_A_Neural_Closed-Loop_Sensor_Simulator_CVPR_2023_paper.html",
      "level": "verified",
      "note": "CVPR 2023. Reconstructs a recorded drive and simulates LiDAR and camera data from new viewpoints. Name collision with UniSim (arXiv 2310.06114), listed under action-video."
     },
     {
      "name": "OccWorld",
      "org": "Tsinghua University",
      "year": 2023,
      "date": "2023-11-27",
      "url": "https://arxiv.org/abs/2311.16038",
      "level": "verified",
      "note": "Predicts future 3D occupancy and the ego car's path for driving; tested on nuScenes. This is a forecasting model, which our current family wording does not cover (see proposed_changes)."
     },
     {
      "name": "WonderWorld",
      "org": "Stanford University; MIT",
      "year": 2024,
      "date": "2024-06-13",
      "url": "https://arxiv.org/abs/2406.09394",
      "level": "verified"
     },
     {
      "name": "HunyuanWorld 1.0",
      "org": "Tencent Hunyuan",
      "year": 2025,
      "date": "2025-07-29",
      "url": "https://arxiv.org/abs/2507.21809",
      "level": "verified"
     },
     {
      "name": "Marble",
      "org": "World Labs",
      "year": 2025,
      "date": "2025-11-12",
      "url": "https://www.worldlabs.ai/blog/marble-world-model",
      "level": "verified",
      "note": "Described by World Labs as generally available for anyone to use."
     }
    ],
    "used_for": [
     "virtual worlds and 3D content",
     "games",
     "testing self-driving software",
     "robot simulation",
     "planning"
    ],
    "sources": [
     {
      "url": "https://openaccess.thecvf.com/content/CVPR2023/html/Yang_UniSim_A_Neural_Closed-Loop_Sensor_Simulator_CVPR_2023_paper.html",
      "title": "UniSim: A Neural Closed-Loop Sensor Simulator",
      "type": "paper",
      "date": "2023-06",
      "accessed": "2026-10-10",
      "note": "Affiliations read from the first page of the open-access PDF (https://openaccess.thecvf.com/content/CVPR2023/papers/Yang_UniSim_A_Neural_Closed-Loop_Sensor_Simulator_CVPR_2023_paper.pdf)."
     },
     {
      "url": "https://arxiv.org/abs/2311.16038",
      "title": "OccWorld: Learning a 3D Occupancy World Model for Autonomous Driving",
      "type": "paper",
      "date": "2023-11-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2406.09394",
      "title": "WonderWorld: Interactive 3D Scene Generation from a Single Image",
      "type": "paper",
      "date": "2024-06-13",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2507.21809",
      "title": "HunyuanWorld 1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/marble-world-model",
      "title": "Marble: A Multimodal World Model",
      "type": "blog",
      "date": "2025-11-12",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.16732v3",
      "title": "A Comprehensive Survey on World Models for Embodied AI",
      "type": "paper",
      "date": "2025-10-19",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section III)."
     },
     {
      "url": "https://arxiv.org/html/2606.00113v1",
      "title": "World Models for Robotic Manipulation: A Survey",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section III-D)."
     }
    ],
    "level": "inferred"
   },
   {
    "id": "learned-simulator",
    "name": "Learned and hybrid physics simulators",
    "plain": "A learned physics simulator is a neural network trained to predict the next physical state of a system, such as the positions and speeds of particles, mesh points or robot joints. It takes the current state and any forces or commands and outputs the state a short time later, and repeating this step produces a full motion. It is used to replace or speed up a traditional physics engine, and in robotics to train controllers inside the learned simulator and to adjust the simulator with real-world data.",
    "how_it_works": [
     "Run a traditional physics simulator, or record real measurements, to collect many examples of a state and the state one time step later.",
     "Describe the system as a graph in which particles, mesh points or robot parts are nodes and nearby or connected pieces are linked.",
     "Train a network that passes messages along the links and predicts how each node moves in the next step.",
     "Apply the network repeatedly to roll the simulation forward for hundreds or thousands of steps. GNS adds noise to its training data to limit the build-up of errors.",
     "In robotics, plug the learned model into a simulator in place of the hand-written dynamics and contact code, train policies there and fine-tune the model with real robot data, as NeRD does."
    ],
    "inputs": [
     "current physical state as particles, mesh points or robot joint positions and speeds",
     "material or object properties",
     "forces or robot commands (optional)"
    ],
    "outputs": [
     "next physical state",
     "a full trajectory when applied repeatedly"
    ],
    "actions": "optional",
    "strengths": [
     "It can run faster than the solver it learned from. MeshGraphNets runs 1-2 orders of magnitude faster than the simulation on which it is trained.",
     "It can handle larger systems than it was trained on. GNS generalises to thousands of timesteps and at least an order of magnitude more particles at test time.",
     "It can learn from real measurements. The NeRD authors report that their learned simulators can be fine-tuned from real-world data, unlike most classical simulators.",
     "The outputs are physical quantities such as positions and speeds, so they can be compared directly with measurements."
    ],
    "limits": [
     "Errors build up over long simulations, so training needs extra measures such as the added noise used in GNS.",
     "Many learned simulators need training for each application and do not transfer to new tasks or environments, as the NeRD authors state.",
     "The examples listed here take the system as particles, meshes or joint states and do not start from raw camera images.",
     "Training examples usually come from a classical simulator, so the learned model can be no more accurate than that data."
    ],
    "inferred_points": [
     "The outputs are physical quantities such as positions and speeds, so they can be compared directly with measurements. (Inferred from the outputs described in the GNS, MeshGraphNets and NeRD papers.)",
     "The examples listed here take the system as particles, meshes or joint states and do not start from raw camera images. (Inferred from the inputs described in the four example papers.)",
     "Training examples usually come from a classical simulator, so the learned model can be no more accurate than that data. (Inferred because GNS and MeshGraphNets describe training on data from the simulator they are compared with.)"
    ],
    "examples": [
     {
      "name": "Interaction Networks",
      "org": "Google DeepMind",
      "year": 2016,
      "date": "2016-12-01",
      "url": "https://arxiv.org/abs/1612.00222",
      "level": "verified",
      "note": "NIPS 2016. Simulated n-body problems, rigid-body collision and non-rigid dynamics."
     },
     {
      "name": "GNS (Learning to Simulate Complex Physics with Graph Networks)",
      "org": "DeepMind; Stanford University",
      "year": 2020,
      "date": "2020-02-21",
      "url": "https://arxiv.org/abs/2002.09405",
      "level": "verified",
      "note": "ICML 2020. Fluids, rigid solids and deformable materials as particles."
     },
     {
      "name": "MeshGraphNets",
      "org": "DeepMind",
      "year": 2020,
      "date": "2020-10-07",
      "url": "https://arxiv.org/abs/2010.03409",
      "level": "verified",
      "note": "ICLR 2021. Aerodynamics, structural mechanics and cloth on meshes."
     },
     {
      "name": "NeRD (Neural Robot Dynamics)",
      "org": "NVIDIA; University of Washington",
      "year": 2025,
      "date": "2025-08-21",
      "url": "https://arxiv.org/abs/2508.15755",
      "level": "verified",
      "note": "Replaces the dynamics and contact solvers inside a robotics simulator; stable and accurate over a thousand simulation steps according to its authors."
     }
    ],
    "used_for": [
     "speeding up physics simulation",
     "science and engineering simulation",
     "robot simulation",
     "training policies"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/1612.00222",
      "title": "Interaction Networks for Learning about Objects, Relations and Physics",
      "type": "paper",
      "date": "2016-12-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2002.09405",
      "title": "Learning to Simulate Complex Physics with Graph Networks",
      "type": "paper",
      "date": "2020-02-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2010.03409",
      "title": "Learning Mesh-Based Simulation with Graph Networks",
      "type": "paper",
      "date": "2020-10-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2508.15755",
      "title": "Neural Robot Dynamics",
      "type": "paper",
      "date": "2025-08-21",
      "accessed": "2026-10-10"
     }
    ],
    "level": "inferred"
   },
   {
    "id": "world-action",
    "status": "proposed",
    "name": "World-action models",
    "why_new": "None of the seven families has robot actions as its main output. The 2026 robotic manipulation survey (integrated prediction-action models), the 2026 agentic world modeling survey (world action models such as DreamZero) and the August 2026 Chef Robotics survey (world-action models) each treat this group separately.",
    "plain": "A world-action model predicts what the scene will look like next and also outputs the actions a robot should take, in one model. It takes camera frames, a task instruction and the robot's state, and outputs robot actions and, in the examples here, predicted future frames as well. It is used directly as the robot's controller.",
    "how_it_works": [
     "Start from a model pretrained to generate video. GR-1 uses large-scale video generative pretraining, and DreamZero builds on a pretrained video diffusion backbone.",
     "Train it on robot data so it predicts future frames and robot actions together.",
     "At run time the robot feeds in its camera view and instruction, and the model outputs the next actions.",
     "Some designs skip drawing the video at run time to save time. UVA decodes video and actions separately so that action output stays fast.",
     "An earlier design generates a video plan first and then extracts actions from it with a separate model. UniPi works this way."
    ],
    "inputs": [
     "camera frames",
     "text instruction",
     "robot state"
    ],
    "outputs": [
     "robot actions",
     "predicted future frames"
    ],
    "actions": "yes",
    "actions_note": "Actions are an output of these models, unlike the other families where actions are an input.",
    "strengths": [
     "NVIDIA reports that DreamZero generalised to new tasks and environments over 2x better than the vision-language-action models it was compared with, in real robot experiments.",
     "Video of other robots or of humans can help. DreamZero reports a relative improvement of over 42% on unseen tasks from 10-20 minutes of video-only demonstrations.",
     "ByteDance Research reports that GR-1 raised the success rate on the CALVIN benchmark from 88.9% to 94.9%."
    ],
    "limits": [
     "Large models are slow. DreamZero needed model and system optimizations to run a 14B model for real-time closed-loop control at 7Hz.",
     "Very fine precision is hard. The DreamZero authors note limitations on tasks requiring sub-centimeter precision, such as key insertion.",
     "The UVA authors state that video-generation-based methods have struggled to match direct policy learning in action accuracy and inference speed.",
     "When prediction is hidden inside the controller, it is hard to tell whether the model learned useful dynamics or only better image features, as the 2026 robotic manipulation survey notes."
    ],
    "inferred_points": [],
    "examples": [
     {
      "name": "UniPi",
      "org": "MIT; Google DeepMind; UC Berkeley; Georgia Tech; University of Alberta",
      "year": 2023,
      "date": "2023-01-31",
      "url": "https://arxiv.org/abs/2302.00111",
      "level": "verified",
      "note": "Generates a video plan from text, then extracts actions from it. The 2606.00113 survey files it under explicit predictive planners."
     },
     {
      "name": "GR-1",
      "org": "ByteDance Research",
      "year": 2023,
      "date": "2023-12-20",
      "url": "https://arxiv.org/abs/2312.13139",
      "level": "verified"
     },
     {
      "name": "UVA (Unified Video Action Model)",
      "org": "Stanford University",
      "year": 2025,
      "date": "2025-02-28",
      "url": "https://arxiv.org/abs/2503.00200",
      "level": "verified"
     },
     {
      "name": "DreamZero",
      "org": "NVIDIA",
      "year": 2026,
      "date": "2026-02-17",
      "url": "https://arxiv.org/abs/2602.15922",
      "level": "verified",
      "note": "The paper calls it a World Action Model (WAM)."
     }
    ],
    "used_for": [
     "robot control"
    ],
    "sources": [
     {
      "url": "https://arxiv.org/abs/2302.00111",
      "title": "Learning Universal Policies via Text-Guided Video Generation",
      "type": "paper",
      "date": "2023-01-31",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2312.13139",
      "title": "Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation",
      "type": "paper",
      "date": "2023-12-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2503.00200",
      "title": "Unified Video Action Model",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2602.15922",
      "title": "World Action Models are Zero-shot Policies",
      "type": "paper",
      "date": "2026-02-17",
      "accessed": "2026-10-10",
      "note": "Limitations discussion read in the HTML full text."
     },
     {
      "url": "https://arxiv.org/html/2606.00113v1",
      "title": "World Models for Robotic Manipulation: A Survey",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section IV-A)."
     },
     {
      "url": "https://arxiv.org/html/2604.22748v3",
      "title": "Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10",
      "note": "Survey by other authors, cited only for its own claims (Section 4.2.1)."
     },
     {
      "url": "https://cdn.prod.website-files.com/64b77d8a4c7f046a3e2e5feb/6a7d12b07356b751a6bcdd08_world_model_survey_0812_2026.pdf",
      "title": "World Models and World-Action Models: An Accessible and Comprehensive Survey",
      "type": "paper",
      "date": "2026-08-12",
      "accessed": "2026-10-10",
      "note": "Preprint by Inkyu Sa (Chef Robotics), marked not peer reviewed; linked from https://chefrobotics.ai/post/world-models-and-world-action-models-an-accessible-and-comprehensive-survey. Cited only for its own classification."
     }
    ],
    "level": "inferred"
   }
  ],
  "survey_comparison": [
   {
    "survey": "A Comprehensive Survey on World Models for Embodied AI",
    "url": "https://arxiv.org/abs/2510.16732",
    "authors": "Xinqing Li, Xin He, Le Zhang, Min Wu, Xiaoli Li, Yun Liu (Nankai University and others)",
    "date": "2025-10-19",
    "version_read": "v3, 2026-06-25",
    "their_classes": [
     "Functionality axis: Decision-Coupled (task-specific) and General-Purpose (task-agnostic) world models",
     "Temporal axis: Sequential Simulation and Inference (one step at a time) and Global Difference Prediction (many future steps at once)",
     "Spatial representation axis: Global Latent Vector; Token Feature Sequence; Spatial Latent Grid (bird's-eye-view or voxel grids); Decomposed Rendering Representation (NeRF, 3D Gaussian splatting)"
    ],
    "maps_to": "The survey sorts each model along three separate axes instead of naming families, so each of our families matches a combination of cells. Our latent-dynamics family matches decision-coupled, step-by-step models with a global latent vector, where the survey lists World Models 2018, PlaNet and Dreamer. Our video-generation family matches general-purpose models that predict many steps at once with token features, where the survey lists Sora. Our action-video family matches general-purpose step-by-step models with token features, where the survey lists iVideoGPT, Genie and Vid2World, and it moves to the decision-coupled side when a video model is tied to one robot task. The survey lists V-JEPA and V-JEPA 2 in the same cell as Sora, so it does not separate feature prediction from pixel prediction at the top level. Our 3d-world family matches two of its representation classes, spatial latent grids (for example the occupancy model OccWorld) and decomposed rendering representations (for example GaussianWorld). The survey has no class for real-time interactive worlds and does not discuss GameNGen or DIAMOND. It also has no class for learned physics simulators, and physics-based models such as PIN-WM and ParticleFormer appear inside its decision-coupled cells.",
    "level": "verified",
    "maps_to_level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.16732",
      "title": "A Comprehensive Survey on World Models for Embodied AI",
      "type": "paper",
      "date": "2025-10-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.16732v3",
      "title": "A Comprehensive Survey on World Models for Embodied AI (HTML, v3)",
      "type": "paper",
      "date": "2026-06-25",
      "accessed": "2026-10-10",
      "note": "Section III (Taxonomy) read in full."
     }
    ]
   },
   {
    "survey": "Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond",
    "url": "https://arxiv.org/abs/2604.22748",
    "authors": "Meng Chu and 49 co-authors",
    "date": "2026-04-24",
    "version_read": "v3, 2026-06-16",
    "their_classes": [
     "Capability levels: L1 Predictor (one-step prediction); L2 Simulator (multi-step, action-conditioned rollouts that respect domain laws); L3 Evolver (revises its own model when predictions fail against new evidence)",
     "Governing-law regimes: physical, digital, social, scientific",
     "Physical-world system types: physics simulation (classical engines); video generation models, split into appearance-first video generation, action-conditioned and interactive video worlds, and decision-oriented video world models; robotics and sim-to-real transfer; spatial reasoning; 3D-structured world models; autonomous driving world models; game world models (placed between the physical and digital regimes)",
     "Architecture axes: representation (symbolic, latent continuous, structured 3D, discrete tokens); dynamics (stochastic latent, deterministic value-aware, autoregressive token, diffusion-based); control interface"
    ],
    "maps_to": "All seven of our families fall inside its physical regime, with games placed between the physical and digital regimes. Its L1 to L3 levels describe how capable a model is and cut across our families, so any of our families can contain L1 or L2 systems. Its group of action-conditioned and interactive video worlds covers both our action-video and interactive-world families, and lists Genie, GAIA-1, Oasis and Matrix-Game 3.0 together. Its appearance-first video generation group (Sora, Lumiere, VideoPoet) matches our video-generation family. Its architecture section separates stochastic latent dynamics (DreamerV3) from deterministic value-aware dynamics (MuZero, TD-MPC2), which are both in our latent-dynamics family, and it groups V-JEPA2 with DreamerV3 as latent continuous representations. Its 3D-structured world models (Marble, RTFM, TesserAct, RoboOccWorld) and its driving occupancy models (OccWorld) match our 3d-world family. Its physics simulation heading covers classical engines such as MuJoCo and Isaac Lab, which it says are not learned simulators, so our learned-simulator family has no direct match; the closest material is its fine-grained physical representations (ParticleFormer, GWM) and its scientific-regime operator learning. Its digital, social and scientific regimes (web and GUI agents, social simulation, AI for science) are outside our scope.",
    "level": "verified",
    "maps_to_level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.22748",
      "title": "Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22748v3",
      "title": "Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond (HTML, v3)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10",
      "note": "Sections 2.4, 4.2.1, 7.1, Appendix B.4 and Appendix C read."
     }
    ]
   },
   {
    "survey": "World Models for Robotic Manipulation: A Survey",
    "url": "https://arxiv.org/abs/2606.00113",
    "authors": "Fangyuan Wang, Ziyuan Wang, Guorui Pei and 15 co-authors",
    "date": "2026-05-27",
    "version_read": "v1, 2026-05-27",
    "their_classes": [
     "Representation families: image and video; learned latent; motion fields and scene flow; geometric and spatiotemporal; physics-informed dynamics",
     "Functional taxonomy: integrated prediction-action models; explicit predictive planners",
     "Infrastructure roles: synthetic experience generation; candidate-action filtering and refinement; search-based action evaluation; learned environments for policy evaluation and improvement; outcome scoring and feasibility verification",
     "Learning lifecycle: pretraining; post-training; inference adaptation"
    ],
    "maps_to": "Its image and video family covers our video-generation and action-video families, and lists UniPi, SuSIE, GR-1, DreamGen, WorldGym and Genie Envisioner there. Its learned latent family covers our latent-dynamics family and also contains V-JEPA 2, so it does not separate our predictive-representation family. Its geometric and spatiotemporal family matches our 3d-world family, with a focus on predicting how a 3D scene changes under robot actions (TesserAct, PointWorld, GWM). Its physics-informed dynamics family overlaps our learned-simulator family and also includes hybrids that add learned parts to differentiable physics (PIN-WM). Its motion fields and scene flow family (FLIP, FlowVLA) has no counterpart in our list. Its integrated prediction-action models (GR-1, GR-2, WorldVLA) have no counterpart either, and we propose a world-action family for them. The survey covers robot manipulation only, so real-time interactive worlds for games are absent. Its infrastructure roles match our used_for values. Synthetic experience generation corresponds to generating training data, and learned environments correspond to evaluating and training policies.",
    "level": "verified",
    "maps_to_level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.00113",
      "title": "World Models for Robotic Manipulation: A Survey",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.00113v1",
      "title": "World Models for Robotic Manipulation: A Survey (HTML, v1)",
      "type": "paper",
      "date": "2026-05-27",
      "accessed": "2026-10-10",
      "note": "Sections II-C, III, IV and V read."
     }
    ]
   }
  ],
  "proposed_changes": [
   "Add an eighth family, world-action models (id world-action), for models that predict future frames and output robot actions in one network, such as GR-1, UVA and DreamZero. None of the current seven families has actions as its main output, and three surveys treat this group separately (arXiv 2606.00113, arXiv 2604.22748 and the August 2026 Chef Robotics survey). A full draft entry is in proposed_families. If you prefer to keep seven families, add a yes/no field outputs_actions instead.",
   "Widen 3d-world to include models that predict how a 3D scene changes over time, such as the driving occupancy model OccWorld, and rename it 3D and 4D world models. All three surveys group occupancy grids, point clouds and Gaussian splats as one kind of representation, while the current wording covers only persistent scenes that can be explored.",
   "Keep predictive-representation but write down its boundary with latent-dynamics. None of the three surveys makes JEPA-style models a separate top-level class. arXiv 2606.00113 and arXiv 2604.22748 group V-JEPA 2 with Dreamer-style latent models, and arXiv 2510.16732 lists V-JEPA next to Sora. The Chef Robotics survey does list JEPA as its own approach. We suggest using predictive-representation when the model learns by predicting features of hidden or future parts and never reconstructs pixels, and using latent-dynamics when the compact state is learned together with rewards or image reconstruction to solve a task. DINO-WM and LeWorldModel sit on the boundary.",
   "Keep action-video and interactive-world separate, because readers care whether a model is for testing machines or for live play, but state the boundary rule. arXiv 2604.22748 groups the two together and arXiv 2510.16732 has no class for interactive worlds. We suggest using interactive-world when the builders report live, frame-by-frame control at a stated frame rate, and using action-video for every other model that takes actions. Under this rule the first Genie model (around 1FPS) is action-video and Genie 3 (24 frames per second) is interactive-world, and the lineage chart can link them.",
   "Rename learned-simulator to Learned and hybrid physics simulators, and state that classical engines such as MuJoCo and Isaac Sim are excluded. arXiv 2606.00113 calls the nearest group physics-informed dynamics and includes hybrids that combine learned parts with differentiable physics, and NeRD replaces only the dynamics and contact solvers inside an existing simulator.",
   "Add an optional sub-type to latent-dynamics with two values. One value is for models that reconstruct observations (World Models 2018, PlaNet, Dreamer), and the other is for value-equivalent models that predict only rewards, values or action choices (MuZero, TD-MPC2). arXiv 2604.22748 separates these as stochastic latent dynamics and deterministic value-aware dynamics, and the Chef Robotics survey lists them as two of its six approaches.",
   "Allow a primary family and an optional secondary family per model. Cosmos is a video generator that NVIDIA also fine-tunes with camera and robot inputs, DIAMOND trains game-playing agents and also runs as a playable Counter-Strike model, and DINO-WM predicts pretrained features and plans actions.",
   "Mention motion and flow predictors (FLIP, FlowVLA), which arXiv 2606.00113 treats as a separate family. We did not research them; we suggest listing them under action-video with a note until more sources treat them as a class.",
   "State on the site that language-based, web, GUI, social and scientific world models are out of scope. arXiv 2604.22748 and the Chef Robotics survey both cover them, so readers who compare our list with those surveys would otherwise see an unexplained gap.",
   "In the lineage chart, draw links across families where a base model is fine-tuned into another family, for example Veo to the Veo-based robot policy evaluator (arXiv 2512.10675) and the Cosmos base models to their robot and driving versions (arXiv 2501.03575, Section 6)."
  ]
 },
 "players": {
  "orgs": [
   {
    "id": "google-deepmind",
    "name": "Google DeepMind",
    "type": "big-tech",
    "region": "europe",
    "hq": "London, United Kingdom (headquarters of Google DeepMind; parent Alphabet is in the United States)",
    "parent": "Alphabet",
    "models": [
     {
      "name": "MuZero",
      "date": "2019-11-19",
      "family": "latent-dynamics",
      "access": "research-only",
      "url": "https://arxiv.org/abs/1911.08265",
      "level": "verified",
      "note": "Plans with a learned model of game dynamics (Atari, Go, chess, shogi)."
     },
     {
      "name": "DreamerV3",
      "date": "2023-01-10",
      "family": "latent-dynamics",
      "access": "open-weights",
      "url": "https://arxiv.org/abs/2301.04104",
      "level": "inferred",
      "note": "General model-based RL agent by Hafner, Pasukonis, Ba and Lillicrap; we did not confirm the authors' affiliations on the abstract page or the licence of the released code, so attribution to Google DeepMind and the access value are our reading."
     },
     {
      "name": "Genie",
      "date": "2024-02-23",
      "family": "action-video",
      "access": "research-only",
      "url": "https://arxiv.org/abs/2402.15391",
      "level": "verified",
      "note": "11B-parameter generative interactive environment trained without action labels on unlabelled Internet videos; learns a latent action space. Second family: interactive-world."
     },
     {
      "name": "Genie 2",
      "date": "2024-12-04",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/",
      "level": "verified",
      "note": "Foundation world model that generates action-controllable, playable 3D environments from a single image, for training and evaluating embodied agents."
     },
     {
      "name": "Veo 2",
      "date": "2024-12-16",
      "family": "video-generation",
      "access": "product",
      "url": "https://blog.google/innovation-and-ai/models-and-research/google-labs/video-image-generation-update-december-2024/",
      "level": "verified"
     },
     {
      "name": "Veo 3",
      "date": "2025-05",
      "family": "video-generation",
      "access": "product",
      "url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Veo-3-Model-Card.pdf",
      "level": "verified",
      "note": "Model card published 2025-05-23; generates video with native audio. Available in the Gemini app, Flow and the Gemini API."
     },
     {
      "name": "Genie 3",
      "date": "2025-08-05",
      "family": "interactive-world",
      "access": "product",
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "level": "verified",
      "note": "Generates navigable worlds from text in real time at 24 frames per second and 720p, consistent for a few minutes. Announced as a limited research preview; later offered to Google AI Ultra subscribers (18+) through the Project Genie prototype (labs.google/projectgenie, launch date not shown on that page). Google DeepMind's SIMA 2 agent was tested in Genie 3 worlds."
     },
     {
      "name": "Veo 3.1",
      "date": "2025-10-15",
      "family": "video-generation",
      "access": "api",
      "url": "https://blog.google/innovation-and-ai/products/veo-updates-flow/",
      "level": "verified",
      "note": "Available in the Gemini app, Flow, the Gemini API and Vertex AI."
     }
    ],
    "technical_position": "Google DeepMind published Genie (2024-02-23), a world model learned from unlabelled Internet video, then Genie 2 (2024-12-04) and Genie 3 (2025-08-05), which generates explorable worlds from text in real time at 24 frames per second and 720p for a few minutes; Genie 3 later became usable by Google AI Ultra subscribers through the Project Genie prototype. It also builds the Veo video generators (Veo 2 on 2024-12-16, Veo 3 in 2025-05, Veo 3.1 on 2025-10-15), and Waymo said on 2026-02-06 that its driving world model is built on Genie 3.",
    "compute_or_data_signals": "Genie: 11B parameters, trained on unlabelled Internet videos (arXiv 2402.15391). Veo 3: trained on audio, video and image data using Google's TPUs with JAX and ML Pathways (model card, 2025-05-23); no data volume or chip count is given. Genie 3: no compute or data figures published on the pages we opened.",
    "funding": [],
    "funding_note": "Alphabet does not disclose spending on world-model research or on Genie and Veo specifically.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/1911.08265",
      "title": "Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model (MuZero)",
      "type": "paper",
      "publisher": "DeepMind (arXiv)",
      "date": "2019-11-19",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2301.04104",
      "title": "Mastering Diverse Domains through World Models (DreamerV3)",
      "type": "paper",
      "publisher": "arXiv (Hafner, Pasukonis, Ba, Lillicrap)",
      "date": "2023-01-10",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2402.15391",
      "title": "Genie: Generative Interactive Environments",
      "type": "paper",
      "publisher": "Google DeepMind (arXiv)",
      "date": "2024-02-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/",
      "title": "Genie 2: A large-scale foundation world model",
      "type": "blog",
      "publisher": "Google DeepMind",
      "date": "2024-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
      "title": "Genie 3: A new frontier for world models",
      "type": "blog",
      "publisher": "Google DeepMind",
      "date": "2025-08-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://deepmind.google/models/genie/",
      "title": "Genie 3 model page",
      "type": "site",
      "publisher": "Google DeepMind",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://labs.google/projectgenie",
      "title": "Project Genie",
      "type": "site",
      "publisher": "Google Labs",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://blog.google/innovation-and-ai/models-and-research/google-labs/video-image-generation-update-december-2024/",
      "title": "State-of-the-art video and image generation with Veo 2 and Imagen 3",
      "type": "blog",
      "publisher": "Google",
      "date": "2024-12-16",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Veo-3-Model-Card.pdf",
      "title": "Veo 3 Model Card",
      "type": "site",
      "publisher": "Google DeepMind",
      "date": "2025-05-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://blog.google/innovation-and-ai/products/veo-updates-flow/",
      "title": "Introducing Veo 3.1 and advanced capabilities in Flow",
      "type": "blog",
      "publisher": "Google",
      "date": "2025-10-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://deepmind.google/models/veo/",
      "title": "Veo model page",
      "type": "site",
      "publisher": "Google DeepMind",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://deepmind.google/blog/sima-2-an-agent-that-plays-reasons-and-learns-with-you-in-virtual-3d-worlds/",
      "title": "SIMA 2: An agent that plays, reasons, and learns with you in virtual 3D worlds",
      "type": "blog",
      "publisher": "Google DeepMind",
      "date": "2025-11-13",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
      "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
      "type": "blog",
      "publisher": "Waymo",
      "date": "2026-02-06",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "openai",
    "name": "OpenAI",
    "type": "big-tech",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Sora",
      "date": "2024-02-15",
      "family": "video-generation",
      "access": "product",
      "url": "https://en.wikipedia.org/wiki/Sora_(text-to-video_model)",
      "level": "reported",
      "note": "Previewed with the technical report 'Video generation models as world simulators'; made public on 2024-12-09 (Wikipedia, citing OpenAI). openai.com returned HTTP 403 to our fetches, so we could not open OpenAI's own pages. Discontinued with the Sora app (see Sora 2)."
     },
     {
      "name": "Sora 2",
      "date": "2025-09-30",
      "family": "video-generation",
      "access": "api",
      "url": "https://developers.openai.com/api/docs/models/sora-2",
      "level": "verified",
      "note": "Unveiled with an iOS app on 2025-09-30 (reported); offered in the API through v1/videos from 2025-10-06 (OpenAI changelog). OpenAI notified developers on 2026-03-24 and shut the Sora 2 models and Videos API down on 2026-09-24 with no replacement (OpenAI docs, verified). Wikipedia, citing OpenAI's help centre, reports the Sora app shut down on 2026-04-26. Current access: none."
     }
    ],
    "technical_position": "OpenAI previewed Sora in 2024-02 with a report framing video generation models as world simulators, released it publicly in 2024-12, and launched Sora 2 with a consumer app on 2025-09-30 and an API on 2025-10-06 (dates of the 2024 and app events are from Wikipedia because openai.com blocked our fetches). OpenAI's API documentation states that the Sora 2 models and Videos API were shut down on 2026-09-24 with no replacement, and Wikipedia reports the Sora app closed on 2026-04-26.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "OpenAI does not disclose spending on Sora or other world-model work. Its general fundraising is not specific to world models and is not listed here.",
    "level": "reported",
    "sources": [
     {
      "url": "https://developers.openai.com/api/docs/models/sora-2",
      "title": "Sora 2 Model | OpenAI API",
      "type": "site",
      "publisher": "OpenAI",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://developers.openai.com/api/docs/deprecations",
      "title": "Deprecations: 2026-03-24 Sora 2 video generation models and Videos API",
      "type": "site",
      "publisher": "OpenAI",
      "date": "2026-03-24",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://developers.openai.com/api/docs/changelog",
      "title": "OpenAI API changelog (Oct 6 entry: v1/videos with Sora 2)",
      "type": "site",
      "publisher": "OpenAI",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://en.wikipedia.org/wiki/Sora_(text-to-video_model)",
      "title": "Sora (text-to-video model)",
      "type": "secondary",
      "publisher": "Wikipedia",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "meta",
    "name": "Meta (FAIR)",
    "type": "big-tech",
    "region": "north-america",
    "hq": "",
    "parent": "Meta Platforms",
    "models": [
     {
      "name": "V-JEPA",
      "date": "2024-02-15",
      "family": "predictive-representation",
      "access": "open-weights",
      "url": "https://ai.meta.com/blog/v-jepa-yann-lecun-ai-model-video-joint-embedding-predictive-architecture/",
      "level": "verified",
      "note": "Video joint-embedding predictive architecture; released under a CC BY-NC licence according to Meta's post."
     },
     {
      "name": "Navigation World Models (NWM)",
      "date": "2024-12-04",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://arxiv.org/abs/2412.03572",
      "level": "verified",
      "note": "Predicts future camera views conditioned on navigation actions; CVPR 2025; code repo created 2025-03-12 (licence not asserted in GitHub metadata). First author affiliation not checked on the abstract page."
     },
     {
      "name": "V-JEPA 2 and V-JEPA 2-AC",
      "date": "2025-06-11",
      "family": "predictive-representation",
      "access": "open-weights",
      "url": "https://ai.meta.com/blog/v-jepa-2-world-model-benchmarks/",
      "level": "verified",
      "note": "1.2 billion parameters; Meta calls it a world model and says code and checkpoints are available for commercial and research use (repo licence MIT). The action-conditioned V-JEPA 2-AC is used for zero-shot robot planning."
     }
    ],
    "technical_position": "Meta's FAIR lab released V-JEPA (2024-02-15) and V-JEPA 2 (2025-06-11), which predict in a learned representation space instead of pixels; V-JEPA 2 is a 1.2 billion-parameter model that Meta calls a world model and uses for zero-shot robot planning after training on 62 hours of robot data. FAIR also published Navigation World Models (2024-12-04), and in 2026-07 Meta Superintelligence Labs announced a separate video generator, Muse Video, which Meta does not describe as a world model.",
    "compute_or_data_signals": "V-JEPA 2: pre-trained on more than 1 million hours of video and 1 million images; the action-conditioned stage used 62 hours of robot data from the DROID dataset; 1.2 billion parameters (Meta, 2025-06-11).",
    "funding": [],
    "funding_note": "Meta does not disclose spending on world-model research. Yann LeCun, who led the JEPA line, left Meta and launched AMI Labs (see AMI Labs).",
    "level": "verified",
    "sources": [
     {
      "url": "https://ai.meta.com/blog/v-jepa-yann-lecun-ai-model-video-joint-embedding-predictive-architecture/",
      "title": "V-JEPA: The next step toward Yann LeCun's vision of advanced machine intelligence (AMI)",
      "type": "blog",
      "publisher": "Meta AI",
      "date": "2024-02-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2404.08471",
      "title": "Revisiting Feature Prediction for Learning Visual Representations from Video (V-JEPA paper)",
      "type": "paper",
      "publisher": "Meta FAIR (arXiv)",
      "date": "2024-02-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://ai.meta.com/blog/v-jepa-2-world-model-benchmarks/",
      "title": "Introducing the V-JEPA 2 world model and new benchmarks for physical reasoning",
      "type": "blog",
      "publisher": "Meta AI",
      "date": "2025-06-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2506.09985",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "publisher": "Meta FAIR (arXiv)",
      "date": "2025-06-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/facebookresearch/vjepa2",
      "title": "facebookresearch/vjepa2",
      "type": "repo",
      "publisher": "Meta FAIR",
      "date": "2025-04-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2412.03572",
      "title": "Navigation World Models",
      "type": "paper",
      "publisher": "Meta FAIR et al. (arXiv)",
      "date": "2024-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/facebookresearch/nwm",
      "title": "facebookresearch/nwm",
      "type": "repo",
      "publisher": "Meta FAIR",
      "date": "2025-03-12",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://ai.meta.com/blog/introducing-muse-image-muse-video-msl/",
      "title": "Introducing Muse Image and Muse Video (Meta Superintelligence Labs)",
      "type": "blog",
      "publisher": "Meta AI",
      "date": "2026-07-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "microsoft",
    "name": "Microsoft (Research and Xbox)",
    "type": "big-tech",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Muse (WHAM, World and Human Action Model)",
      "date": "2025-02-19",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://www.nature.com/articles/s41586-025-08600-3",
      "level": "verified",
      "note": "Predicts game visuals and controller actions for the Xbox game Bleeding Edge (Ninja Theory). Weights for 200M and 1.6B versions are on Hugging Face under the Microsoft Research License (non-commercial research use only)."
     },
     {
      "name": "WHAMM (World and Human Action MaskGIT Model)",
      "date": "2025-04-04",
      "family": "interactive-world",
      "access": "product",
      "url": "https://www.microsoft.com/en-us/research/articles/whamm-real-time-world-modelling-of-interactive-environments/",
      "level": "verified",
      "note": "Real-time playable AI rendition of Quake II at 10+ frames a second, offered as an experience in Copilot Labs; trained on 1 week of gameplay data. We did not confirm that the Copilot Labs demo is still online."
     }
    ],
    "technical_position": "Microsoft Research and Xbox studio Ninja Theory published WHAM (branded Muse) in Nature on 2025-02-19, a model trained on human gameplay that generates game visuals and controller actions, and released its weights for non-commercial research. On 2025-04-04 Microsoft showed WHAMM, a real-time playable Quake II world model in Copilot Labs; we found no later Microsoft world-model release on the pages we could open.",
    "compute_or_data_signals": "WHAM: trained on about 500,000 anonymized Bleeding Edge sessions (over 7 years of continuous play) for the 7 Maps dataset; the released model card describes one year of gameplay from 27,990 players; 1.6B-parameter model evaluated in the paper. WHAMM: 1 week of gameplay data.",
    "funding": [],
    "funding_note": "Microsoft does not disclose spending on world-model research. microsoft.com research pages returned HTTP 403 to direct fetches, so we could not check for 2026 releases.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.nature.com/articles/s41586-025-08600-3",
      "title": "World and Human Action Models towards gameplay ideation (Nature)",
      "type": "paper",
      "publisher": "Nature / Microsoft Research",
      "date": "2025-02-19",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/microsoft/wham",
      "title": "microsoft/wham model card",
      "type": "repo",
      "publisher": "Microsoft Research",
      "date": "2025-02-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/microsoft/wham/blob/main/LICENSE.md",
      "title": "Microsoft Research License Terms (WHAM)",
      "type": "repo",
      "publisher": "Microsoft",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.microsoft.com/en-us/research/articles/whamm-real-time-world-modelling-of-interactive-environments/",
      "title": "WHAMM! Real-time world modelling of interactive environments",
      "type": "blog",
      "publisher": "Microsoft Research",
      "date": "2025-04-04",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "nvidia",
    "name": "NVIDIA",
    "type": "big-tech",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Cosmos (Predict1, Transfer1, Reason1)",
      "date": "2025-01-07",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://arxiv.org/abs/2501.03575",
      "level": "verified",
      "note": "World foundation model platform; arXiv v1 2025-01-07; code repos for Predict1, Transfer1 and Reason1 under Apache-2.0 (created 2025-03-02). Predict models also take actions after post-training (second family: action-video)."
     },
     {
      "name": "Cosmos-Predict2",
      "date": "2025-06-11",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/nvidia-cosmos/cosmos-predict2",
      "level": "inferred",
      "note": "Date is the GitHub repository creation date (Apache-2.0 code)."
     },
     {
      "name": "Cosmos-Predict2.5 and Transfer2.5",
      "date": "2025-10-28",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://arxiv.org/abs/2511.00062",
      "level": "verified",
      "note": "Unifies Text2World, Image2World and Video2World in one flow-based model; released at 2B and 14B. Repos created 2025-09-25."
     },
     {
      "name": "Cosmos 3 (Nano, Super)",
      "date": "2026-06-01",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://arxiv.org/abs/2606.02800",
      "level": "verified",
      "note": "Omnimodal world models that process and generate language, image, video, audio and action; code, checkpoints and data released under the Linux Foundation OpenMDW-1.1 licence. Second family: video-generation."
     }
    ],
    "technical_position": "NVIDIA released the Cosmos world foundation model platform with open weights in 2025-01 and followed with Cosmos-Predict2 (2025-06), Cosmos-Predict2.5 (2025-10) and Cosmos 3 (2026-06-01), an omnimodal model family that generates video, audio and robot actions. NVIDIA positions Cosmos as an open base that robot and driving developers fine-tune into their own world models.",
    "compute_or_data_signals": "Cosmos (2025-01): about 20M hours of raw video collected, from which about 100M clips of 2 to 60 seconds were extracted. Cosmos-Predict2.5: trained on 200M curated video clips; 2B and 14B models. Cosmos 3: Cosmos3-Nano pre-trained on 31.05T tokens using 1024 NVIDIA GB200 GPUs; Cosmos3-Super on 17.86T tokens using 2048 GB200 GPUs (arXiv 2606.02800).",
    "funding": [],
    "funding_note": "NVIDIA does not disclose its spending on Cosmos. NVIDIA and its venture arm NVentures appear as investors in several world-model start-ups in this file; those investments are recorded under each start-up.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2501.03575",
      "title": "Cosmos World Foundation Model Platform for Physical AI",
      "type": "paper",
      "publisher": "NVIDIA (arXiv)",
      "date": "2025-01-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2501.03575v3",
      "title": "Cosmos World Foundation Model Platform for Physical AI (HTML v3)",
      "type": "paper",
      "publisher": "NVIDIA (arXiv)",
      "date": "2025-07-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/nvidia-cosmos",
      "title": "nvidia-cosmos GitHub organisation (repository list)",
      "type": "repo",
      "publisher": "NVIDIA",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2511.00062",
      "title": "World Simulation with Video Foundation Models for Physical AI (Cosmos-Predict2.5)",
      "type": "paper",
      "publisher": "NVIDIA (arXiv)",
      "date": "2025-10-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2606.02800",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "paper",
      "publisher": "NVIDIA (arXiv)",
      "date": "2026-06-01",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3 (HTML v4)",
      "type": "paper",
      "publisher": "NVIDIA (arXiv)",
      "date": "2026-06-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.nvidia.com/en-us/ai/cosmos/",
      "title": "NVIDIA Cosmos product page",
      "type": "site",
      "publisher": "NVIDIA",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "tesla",
    "name": "Tesla",
    "type": "big-tech",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [],
    "technical_position": "Tesla builds end-to-end driving software (FSD) and the Optimus humanoid, and its executives have described a learned world simulator in conference talks and social posts. We could not open a primary Tesla source describing a world model: tesla.com and ir.tesla.com returned HTTP 403, and Tesla's quarterly updates from Q2 2025 to Q2 2026 and its 2025 10-K, which we read on SEC EDGAR, do not mention a world model or world simulator.",
    "compute_or_data_signals": "Company-wide AI training compute, not specific to any world model: Cortex reached 67k H100 equivalents (Q2 2025 update) and 81k H100 equivalents (Q3 2025 update); Cortex 1 >100k H100e and Cortex 2 >130k H100e in early ramp (Q1 2026 update); Cortex 1 >90 MW and Cortex 2 >115 MW (Q2 2026 update).",
    "funding": [],
    "funding_note": "Tesla is a listed company and does not disclose spending on any world simulator.",
    "level": "unknown",
    "sources": [
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828025035738/exhibit991.htm",
      "title": "Tesla Q2 2025 Update (8-K Exhibit 99.1)",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2025-07-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828025045861/exhibit991.htm",
      "title": "Tesla Q3 2025 Update (8-K Exhibit 99.1)",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2025-10-22",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828026003837/exhibit991.htm",
      "title": "Tesla Q4 and FY 2025 Update (8-K Exhibit 99.1)",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2026-01-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828026026551/exhibit991.htm",
      "title": "Tesla Q1 2026 Update (8-K Exhibit 99.1)",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2026-04-22",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828026049213/exhibit991.htm",
      "title": "Tesla Q2 2026 Update (8-K Exhibit 99.1)",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2026-07-22",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.sec.gov/Archives/edgar/data/1318605/000162828026003952/tsla-20251231.htm",
      "title": "Tesla Form 10-K for fiscal year 2025",
      "type": "filing",
      "publisher": "Tesla (SEC EDGAR)",
      "date": "2026-01-29",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "waymo",
    "name": "Waymo",
    "type": "robot-company",
    "region": "north-america",
    "hq": "",
    "parent": "Alphabet",
    "models": [
     {
      "name": "Waymo World Model",
      "date": "2026-02-06",
      "family": "action-video",
      "access": "internal",
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
      "level": "verified",
      "note": "Generative model built on Google DeepMind's Genie 3 and adapted for driving; generates camera and lidar data controlled by language prompts, driving inputs and scene layouts, used to simulate rare events for the Waymo Driver. Second family: learned-simulator."
     }
    ],
    "technical_position": "On 2026-02-06 Waymo introduced the Waymo World Model, built on Google DeepMind's Genie 3, which generates camera and lidar data for driving simulation and can be steered by text prompts, driving inputs and scene layouts. Waymo says its driver has driven billions of miles in virtual worlds alongside nearly 200 million fully autonomous real miles.",
    "compute_or_data_signals": "Waymo states nearly 200 million fully autonomous miles driven by 2026-02-06 and billions of simulated miles; no training compute figures for the world model are published.",
    "funding": [
     {
      "date": "2024-10",
      "round": "Investment round",
      "amount": "$5.6 billion",
      "valuation": "",
      "lead_investors": [
       "Alphabet"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://waymo.com/blog/2024/10/investing-to-bring-the-waymo-driver-to-more-riders/",
        "title": "Investing to bring the Waymo Driver to more riders",
        "type": "blog",
        "publisher": "Waymo",
        "date": "2024-10-25",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Continued participation from Andreessen Horowitz, Fidelity, Perry Creek, Silver Lake, Tiger Global and T. Rowe Price."
     },
     {
      "date": "2026-02",
      "round": "Investment round",
      "amount": "$16 billion",
      "valuation": "",
      "lead_investors": [
       "Dragoneer Investment Group",
       "DST Global",
       "Sequoia Capital"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://waymo.com/blog/2026/02/waymo-raises-usd16-billion-investment-round/",
        "title": "Accelerating our global growth: Waymo raises $16 billion investment round",
        "type": "blog",
        "publisher": "Waymo",
        "date": "2026-02-02",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Alphabet is the majority investor; significant investments from Andreessen Horowitz and Mubadala Capital, plus Bessemer, Silver Lake, Tiger Global, T. Rowe Price and Temasek, among others. Waymo's post does not state a valuation."
     },
     {
      "date": "2026-10",
      "round": "Debt financing (term loan)",
      "amount": "$5 billion",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://waymo.com/blog/2026/10/waymo-closes-5-billion-debt-financing/",
        "title": "Waymo Closes $5 Billion Debt Financing to Accelerate Business Expansion",
        "type": "blog",
        "publisher": "Waymo",
        "date": "2026-10-08",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Waymo's first debt financing."
     }
    ],
    "funding_note": "Waymo does not publish valuations in its own posts, and funding is for the whole company, not the world model.",
    "level": "verified",
    "sources": [
     {
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/",
      "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
      "type": "blog",
      "publisher": "Waymo",
      "date": "2026-02-06",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "wayve",
    "name": "Wayve",
    "type": "start-up",
    "region": "europe",
    "hq": "London, United Kingdom (company investors page)",
    "parent": "",
    "models": [
     {
      "name": "GAIA-1",
      "date": "2023-09-29",
      "family": "action-video",
      "access": "research-only",
      "url": "https://wayve.ai/press/wayve-releases-gaia-1-technical-report/",
      "level": "verified",
      "note": "9-billion-parameter generative world model for driving that takes video, text and action inputs; technical report dated 2023-09-29 on the GAIA page and announced 2023-10-03."
     },
     {
      "name": "GAIA-2",
      "date": "2025-03-26",
      "family": "action-video",
      "access": "internal",
      "url": "https://wayve.ai/press/wayve-unveils-gaia2/",
      "level": "verified",
      "note": "Controllable multi-view generative world model used for synthetic data and validation of Wayve's driving system."
     },
     {
      "name": "GAIA-3",
      "date": "2025-12-02",
      "family": "action-video",
      "access": "internal",
      "url": "https://wayve.ai/press/wayve-launches-gaia3/",
      "level": "verified",
      "note": "15 billion parameters, double GAIA-2, pre-trained on ten times more data; used to evaluate and validate driving AI. Wayve works with WMG, University of Warwick, on validating such models for safety evaluation."
     },
     {
      "name": "GAIA-4",
      "date": "2026-08-03",
      "family": "action-video",
      "access": "internal",
      "url": "https://wayve.ai/thinking/gaia-4/",
      "level": "verified",
      "note": "Multimodal world model (camera, radar and lidar generated together) for closed-loop simulation with the Wayve AI Driver in the loop."
     }
    ],
    "technical_position": "Wayve has published a series of driving world models: GAIA-1 (9 billion parameters, 2023-09), GAIA-2 (2025-03-26), GAIA-3 (15 billion parameters, 2025-12-02) and GAIA-4 (2026-08-03), which generates camera, radar and lidar data together for closed-loop testing of its driving model. The company describes these models as tools to generate synthetic data and to evaluate and validate its end-to-end AI Driver.",
    "compute_or_data_signals": "GAIA-1: 9 billion parameters, trained on approximately 4,700 hours of Wayve's UK driving data (Wayve, 2023-10-03). GAIA-3: 15 billion parameters, twice GAIA-2, with a video tokenizer twice as large and ten times more pre-training data than GAIA-2 (Wayve, 2025-12-02).",
    "funding": [
     {
      "date": "2022-01",
      "round": "Series B",
      "amount": "$200 million",
      "valuation": "",
      "lead_investors": [
       "Eclipse"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/wayve-announces-200-million-in-funding-to-accelerate-the-development-of-av2-0-the-next-wave-of-autonomous-vehicles/",
        "title": "Wayve announces $200 Million in Funding to Accelerate AV2.0",
        "type": "press-release",
        "publisher": "Wayve",
        "date": "2022-01-18",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from D1 Capital Partners, Baillie Gifford, Moore Strategic Ventures, Linse Capital, Microsoft, Virgin, Compound and Balderton; total equity raised over $258 million."
     },
     {
      "date": "2024-05",
      "round": "Series C",
      "amount": "$1.05 billion",
      "valuation": "",
      "lead_investors": [
       "SoftBank Group"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/series-c/",
        "title": "Wayve Raises Over $1 Billion Led by SoftBank to Develop Embodied AI Products for Automated Driving",
        "type": "press-release",
        "publisher": "Wayve",
        "date": "2024-05-07",
        "accessed": "2026-10-11"
       }
      ],
      "note": "New investor NVIDIA and existing investor Microsoft contributed."
     },
     {
      "date": "2026-02",
      "round": "Series D",
      "amount": "$1.2B (the company headline says it secured $1.5B including additional milestone-based capital from Uber)",
      "valuation": "$8.6 billion post-money",
      "lead_investors": [
       "Eclipse",
       "Balderton",
       "SoftBank Vision Fund 2"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/series-d/",
        "title": "Wayve secures $1.5B to deploy its global autonomy platform",
        "type": "press-release",
        "publisher": "Wayve",
        "date": "2026-02-25",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from Microsoft, NVIDIA and Uber and automakers Mercedes-Benz, Nissan and Stellantis; new investors include Ontario Teachers' Pension Plan, Baillie Gifford and British Business Bank."
     },
     {
      "date": "2026-04",
      "round": "Series D extension",
      "amount": "$60M",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/series-d-extension/",
        "title": "Wayve Broadens Silicon Backing with $60M Investment from AMD, Arm and Qualcomm",
        "type": "press-release",
        "publisher": "Wayve",
        "date": "2026-04-15",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Investment from AMD, Arm and Qualcomm."
     }
    ],
    "funding_note": "Wayve's investors page states $2.8B total funding in 4 rounds. Its 2026-07-01 employee tender offer of $85 million is a secondary sale, not new funding.",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/labs/gaia/",
      "title": "GAIA: generative world models for autonomy (Wayve Labs page)",
      "type": "site",
      "publisher": "Wayve",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/press/wayve-releases-gaia-1-technical-report/",
      "title": "Wayve releases GAIA-1 technical report",
      "type": "press-release",
      "publisher": "Wayve",
      "date": "2023-10-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/press/wayve-unveils-gaia2/",
      "title": "Wayve unveils GAIA-2",
      "type": "press-release",
      "publisher": "Wayve",
      "date": "2025-03-26",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/press/wayve-launches-gaia3/",
      "title": "Wayve launches GAIA-3, advancing world models from simulation to evaluation",
      "type": "press-release",
      "publisher": "Wayve",
      "date": "2025-12-02",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/thinking/gaia-4/",
      "title": "GAIA-4: Multimodal World Models Powering Closed-Loop Simulation for Safe and Scalable Autonomy",
      "type": "blog",
      "publisher": "Wayve",
      "date": "2026-08-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/company/investors/",
      "title": "Wayve investors page",
      "type": "site",
      "publisher": "Wayve",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "waabi",
    "name": "Waabi",
    "type": "start-up",
    "region": "north-america",
    "hq": "Toronto, Canada (not confirmed on the pages we opened)",
    "parent": "",
    "models": [
     {
      "name": "Waabi World",
      "date": "2022-02-09",
      "family": "learned-simulator",
      "access": "internal",
      "url": "https://waabi.ai/insights/waabi-world",
      "level": "verified",
      "note": "Waabi calls it a neural simulator: it generates sensor data and AI-driven actors to train and test the Waabi Driver; an onboard version runs in a few milliseconds for mixed-reality testing on real trucks (2025-07-14). The brief's learned-simulator family is about physics engines; this is a learned driving and sensor simulator."
     },
     {
      "name": "UniSim",
      "date": "2023-06-14",
      "family": "3d-world",
      "access": "research-only",
      "url": "https://waabi.ai/insights/introducing-unisim-one-of-the-core-groundbreaking-technologies-powering-waabi-world",
      "level": "verified",
      "note": "Neural closed-loop sensor simulator presented at CVPR 2023 (with University of Toronto authors); re-renders camera and LiDAR data for new trajectories."
     },
     {
      "name": "Copilot4D",
      "date": "2024-03-15",
      "family": "action-video",
      "access": "research-only",
      "url": "https://waabi.ai/insights/introducing-copilot4d",
      "level": "verified",
      "note": "World model that forecasts future LiDAR point clouds (not video) conditioned on the vehicle's actions; ICLR 2024."
     }
    ],
    "technical_position": "Waabi builds autonomous trucking and robotaxi software around Waabi World, a neural simulator it introduced on 2022-02-09, and has published UniSim (2023-06-14, a neural sensor simulator) and Copilot4D (2024-03-15, a LiDAR world model). In 2025-07 it described running an onboard version of Waabi World on its trucks to insert virtual actors into real sensor data during testing.",
    "compute_or_data_signals": "",
    "funding": [
     {
      "date": "2023-01",
      "round": "Strategic investment",
      "amount": "not disclosed",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://waabi.ai/insights/welcoming-volvo-group-venture-capital-as-a-strategic-investor-in-waabi",
        "title": "Welcoming Volvo Group Venture Capital as a strategic investor in Waabi",
        "type": "blog",
        "publisher": "Waabi",
        "date": "2023-01-18",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Volvo Group Venture Capital became a strategic investor."
     },
     {
      "date": "2024-06",
      "round": "Series B",
      "amount": "$200 million (USD)",
      "valuation": "",
      "lead_investors": [
       "Uber",
       "Khosla Ventures"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://waabi.ai/insights/waabi-series-b-announcement",
        "title": "Waabi raises $200M to launch fully driverless trucks in 2025",
        "type": "press-release",
        "publisher": "Waabi",
        "date": "2024-06-18",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from NVIDIA, Volvo Group Venture Capital, Porsche Automobil Holding SE, Scania Invest and Ingka Investments."
     },
     {
      "date": "2026-01",
      "round": "Series C",
      "amount": "$750M USD (the company headline says $1 Billion including a milestone-based future investment from Uber)",
      "valuation": "",
      "lead_investors": [
       "Khosla Ventures",
       "G2 Venture Partners"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://waabi.ai/insights/waabi-secures-1-billion-in-new-funding-to-lead-physical-ai-revolution",
        "title": "Waabi secures $1 Billion in new funding to lead Physical AI revolution",
        "type": "press-release",
        "publisher": "Waabi",
        "date": "2026-01-28",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "funding_note": "Waabi has not published valuations. Its 2021 Series A is not listed because the search budget ran out before we could open a source for it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://waabi.ai/press",
      "title": "Waabi press and insights index",
      "type": "site",
      "publisher": "Waabi",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://waabi.ai/insights/waabi-world",
      "title": "Waabi World",
      "type": "blog",
      "publisher": "Waabi",
      "date": "2022-02-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://waabi.ai/insights/introducing-unisim-one-of-the-core-groundbreaking-technologies-powering-waabi-world",
      "title": "Introducing UniSim, one of the core technologies powering Waabi World",
      "type": "blog",
      "publisher": "Waabi",
      "date": "2023-06-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://waabi.ai/insights/introducing-copilot4d",
      "title": "Introducing Copilot4D",
      "type": "blog",
      "publisher": "Waabi",
      "date": "2024-03-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://waabi.ai/insights/mixed-reality-testing-pushes-the-boundaries-of-av-safety",
      "title": "The ultimate driving test for AI: Mixed Reality Testing pushes the boundaries of AV safety",
      "type": "blog",
      "publisher": "Waabi",
      "date": "2025-07-14",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "1x",
    "name": "1X Technologies",
    "type": "robot-company",
    "region": "multi",
    "hq": "Founded in Norway (formerly Halodi Robotics); the company site does not state its current headquarters",
    "parent": "",
    "models": [
     {
      "name": "1X World Model (first release)",
      "date": "2024-09-17",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://www.1x.tech/discover/1x-world-model",
      "level": "verified",
      "note": "Action-conditioned video model trained on thousands of hours of EVE humanoid data, used to evaluate robot policies. 1X released over 100 hours of vector-quantized video (Apache 2.0), baseline models with pretrained weights (Llama- and GENIE-based) on GitHub, and the 1X World Model Challenge; 100 hours of raw video followed on 2024-11-05 under CC-BY-NC-SA 4.0. The weights released are baselines; we did not confirm whether they are the same as 1X's internal model."
     },
     {
      "name": "1XWM (policy evaluation model)",
      "date": "2025-06-16",
      "family": "action-video",
      "access": "internal",
      "url": "https://www.1x.tech/discover/redwood-ai-world-model",
      "level": "verified",
      "note": "Predicts outcomes of NEO's actions to evaluate Redwood AI policy checkpoints; 1X reports a correlation between predicted and real success rates. A paper is linked from the post."
     },
     {
      "name": "1XWM as robot policy",
      "date": "2026-01-12",
      "family": "world-action",
      "access": "internal",
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "level": "verified",
      "note": "Video-pretrained world model integrated into NEO as a policy: actions are derived from text-conditioned video generation."
     }
    ],
    "technical_position": "1X released a robot world model with a public dataset and challenge on 2024-09-17, used its world model (1XWM) to predict real-world success rates of NEO policies by 2025-06-16, and on 2026-01-12 described 1XWM running on NEO as a policy that derives actions from generated video. On 2026-06-04 it set up the 1X World Model Lab to scale embodied world-model pretraining.",
    "compute_or_data_signals": "1X World Model (2024-09-17): trained on thousands of hours of EVE humanoid data; over 100 hours of tokenized video released under Apache 2.0, plus 100 hours of raw video (2024-11-05) under CC-BY-NC-SA 4.0. 1XWM policy (2026-01-12): mid-trained on 900 hours of egocentric human video; post-training data is 98.5% pick-and-place.",
    "funding": [
     {
      "date": "2023-03",
      "round": "Series A2",
      "amount": "$23.5 million",
      "valuation": "",
      "lead_investors": [
       "OpenAI Startup Fund"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/1x-rasies-23-5m-in-series-a2-funding-led-by-open-ai",
        "title": "1X Raises $23.5M in Series A2 Funding led by OpenAI",
        "type": "press-release",
        "publisher": "1X Technologies",
        "date": "2023-03-23",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from Tiger Global and Norway-based investors including Sandwater, Alliance Ventures and Skagerak Capital. The company then operated as Halodi Robotics until its rename to 1X."
     },
     {
      "date": "2024-01",
      "round": "Series B",
      "amount": "$100 million",
      "valuation": "",
      "lead_investors": [
       "EQT Ventures"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/1x-secures-100m-in-series-b-funding",
        "title": "Series B: 1X Secures $100M Funding",
        "type": "press-release",
        "publisher": "1X Technologies",
        "date": "2024-01",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://eqtgroup.com/thinq/early-stage/norwegian-robotics-startup-1x-secures-100m-in-series-b-funding-led-by-eqt-ventures",
        "title": "Norwegian Robotics Startup 1X Secures $100M in Series B Funding Led by EQT Ventures",
        "type": "blog",
        "publisher": "EQT (ThinQ)",
        "date": "2024-06-08",
        "accessed": "2026-10-11"
       }
      ],
      "note": "1X's own release (dated JAN 13 '24 on its site) says EQT Ventures participated; EQT's ThinQ article (dated 2024-06-08, a later republication) says the round was led by EQT Ventures, so the lead is marked from the investor's statement. 1X says it raised over $125 million in under 12 months. A post-money valuation of $820 million appears only in data aggregators."
     }
    ],
    "funding_note": "We found no announced round after 2024-01. News reports in 2025 and 2026 describe talks (up to $1 billion at about $10 billion, and SoftBank discussing a controlling stake at about $6 billion) that we could not confirm as closed.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/1x-world-model",
      "title": "1X World Model",
      "type": "blog",
      "publisher": "1X Technologies",
      "date": "2024-09-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/discover/1x-world-model-sampling-challenge",
      "title": "1X World Model: Sampling Challenge Update",
      "type": "blog",
      "publisher": "1X Technologies",
      "date": "2024-11-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/discover/redwood-ai-world-model",
      "title": "1X World Model (policy evaluation)",
      "type": "blog",
      "publisher": "1X Technologies",
      "date": "2025-06-16",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/discover/world-model-self-learning",
      "title": "1X World Model | From Video to Action: A New Way Robots Learn",
      "type": "blog",
      "publisher": "1X Technologies",
      "date": "2026-01-12",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/discover/1x-world-model-lab",
      "title": "The 1X World Model Lab | Est. 2026",
      "type": "blog",
      "publisher": "1X Technologies",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "physical-intelligence",
    "name": "Physical Intelligence",
    "type": "robot-company",
    "region": "north-america",
    "hq": "San Francisco, United States (CNBC)",
    "parent": "",
    "models": [],
    "technical_position": "Physical Intelligence builds vision-language-action robot policies: π0 (2024-10-31, weights released 2025-02-04), π0.5 (2025-04-22), π*0.6 (2025-11-17) and π0.7 (2026-04-16). Its blog index as archived on 2026-10-03 lists no world model release, and we found none elsewhere.",
    "compute_or_data_signals": "",
    "funding": [
     {
      "date": "2024-03",
      "round": "Seed",
      "amount": "$70 million (reported)",
      "valuation": "$400 million (reported)",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://www.nbcnewyork.com/news/business/money-report/jeff-bezos-and-openai-invest-in-robot-startup-physical-intelligence-at-2-4-billion-valuation/5952538/?amp=1",
        "title": "Jeff Bezos and OpenAI invest in robot startup Physical Intelligence at $2.4 billion valuation (CNBC)",
        "type": "secondary",
        "publisher": "CNBC via NBC New York",
        "date": "2024-11-04",
        "accessed": "2026-10-11"
       }
      ],
      "note": "CNBC says the March seed reportedly came in at these terms; the company did not confirm them in that article."
     },
     {
      "date": "2024-11",
      "round": "Not named",
      "amount": "$400 million",
      "valuation": "$2.4 billion post-money",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://www.nbcnewyork.com/news/business/money-report/jeff-bezos-and-openai-invest-in-robot-startup-physical-intelligence-at-2-4-billion-valuation/5952538/?amp=1",
        "title": "Jeff Bezos and OpenAI invest in robot startup Physical Intelligence at $2.4 billion valuation (CNBC)",
        "type": "secondary",
        "publisher": "CNBC via NBC New York",
        "date": "2024-11-04",
        "accessed": "2026-10-11"
       }
      ],
      "note": "The company confirmed the round to CNBC; investors named by a spokesperson: Jeff Bezos, OpenAI, Thrive Capital and Lux Capital. We found no announcement on the company's own site."
     },
     {
      "date": "2025-11",
      "round": "Not named",
      "amount": "$600 million",
      "valuation": "$5.6 billion (reported)",
      "lead_investors": [
       "CapitalG"
      ],
      "level": "reported",
      "sources": [
       {
        "url": "https://siliconangle.com/2025/11/20/jeff-bezos-backed-physical-intelligence-raises-600m-improve-ai-robot-brains/",
        "title": "Jeff Bezos-backed Physical Intelligence raises $600M to improve AI robot brains",
        "type": "secondary",
        "publisher": "SiliconANGLE",
        "date": "2025-11-20",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Based on The Information and Bloomberg reports; returning investors Lux Capital, Thrive Capital and Jeff Bezos; new investors Index Ventures and T. Rowe Price. The company did not officially disclose the round."
     }
    ],
    "funding_note": "Physical Intelligence has not announced its rounds on its own site. Bloomberg reported on 2026-03-27 that it was discussing about $1 billion at more than $11 billion; we found no confirmation that this round closed.",
    "level": "verified",
    "sources": [
     {
      "url": "https://web.archive.org/web/20261003065239/https://www.pi.website/blog",
      "title": "Physical Intelligence (π) – Blog (archived copy, 2026-10-03; live site returned HTTP 429)",
      "type": "blog",
      "publisher": "Physical Intelligence",
      "date": "2026-10",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "skild-ai",
    "name": "Skild AI",
    "type": "robot-company",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [],
    "technical_position": "Skild AI builds a robot foundation model it calls the Skild Brain, trained on large-scale simulation, human videos and teleoperation, and described an omni-bodied version on 2025-09-24 and an in-context learning model named S1 in 2026. We found no released or demonstrated world model on its blog as of 2026-10-11.",
    "compute_or_data_signals": "Skild AI's Series C post says its pretraining uses trillions of synthetic simulated experiences and billions of human action videos (2026-01-14); these figures describe its robot policy data, not a world model.",
    "funding": [
     {
      "date": "2024-07",
      "round": "Series A",
      "amount": "$300 million",
      "valuation": "$1.5 billion",
      "lead_investors": [
       "Lightspeed Venture Partners",
       "Coatue",
       "SoftBank Group",
       "Bezos Expeditions"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.skild.ai/blogs/announcing-our-300m-series-a",
        "title": "Announcing our $300M Series A Funding",
        "type": "blog",
        "publisher": "Skild AI",
        "date": "2024-07-09",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from Felicis Ventures, Sequoia, Menlo Ventures, General Catalyst, CRV and others."
     },
     {
      "date": "2025",
      "round": "Series B",
      "amount": "$500 million (reported)",
      "valuation": "$4.7 billion (reported)",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://www.marketscreener.com/news/softbank-nvidia-looking-to-invest-in-skild-ai-at-14-billion-valuation-sources-say-ce7d51ddd08ef024",
        "title": "SoftBank, Nvidia looking to invest in Skild AI at $14 billion valuation, sources say (Reuters)",
        "type": "secondary",
        "publisher": "Reuters via MarketScreener",
        "date": "2025-12-08",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Reuters, citing PitchBook data, with participation from Nvidia, LG's venture arm and Samsung. Other reports give $100 million (SoftBank's share), $200 million or $135 million and a $4.5 billion valuation; we found no company announcement."
     },
     {
      "date": "2026-01",
      "round": "Series C",
      "amount": "$1.4 billion",
      "valuation": "over $14 billion",
      "lead_investors": [
       "SoftBank"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.skild.ai/blogs/series-c",
        "title": "Announcing Series C",
        "type": "blog",
        "publisher": "Skild AI",
        "date": "2026-01-14",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Participation from NVentures (NVIDIA), Macquarie Capital and Jeff Bezos (via Bezos Expeditions); returning Lightspeed, Felicis, Coatue and Sequoia Capital; strategic investors including LG and Schneider Electric."
     }
    ],
    "funding_note": "",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.skild.ai/blogs",
      "title": "Skild AI blog index",
      "type": "blog",
      "publisher": "Skild AI",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.skild.ai/blogs/omni-bodied",
      "title": "The case for an omni-bodied robot brain",
      "type": "blog",
      "publisher": "Skild AI",
      "date": "2025-09-24",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.skild.ai/blogs/s1",
      "title": "Introducing S1: In-Context Learning for Robotics",
      "type": "blog",
      "publisher": "Skild AI",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.skild.ai/blogs/series-c",
      "title": "Announcing Series C",
      "type": "blog",
      "publisher": "Skild AI",
      "date": "2026-01-14",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "figure-ai",
    "name": "Figure AI",
    "type": "robot-company",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [],
    "technical_position": "Figure builds humanoid robots and the Helix robot policy family (Helix 02 on 2026-01-27, Helix 2.5 on 2026-09-17) trained increasingly on human video, including its Index dataset announced on 2026-08-25. None of the Figure news posts we checked describes a world model, and we found no world model release as of 2026-10-11.",
    "compute_or_data_signals": "Figure says it has committed $3.5B of compute to training Helix (2026-09-17) and that its Index app has received over 16M videos and is processing 30 minutes of video uploads every second (2026-08-25). These figures describe policy training data, not a world model.",
    "funding": [
     {
      "date": "2025-09",
      "round": "Series C",
      "amount": "more than $1 billion in committed capital",
      "valuation": "$39 billion post-money",
      "lead_investors": [
       "Parkway Venture Capital"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.figure.ai/news/series-c",
        "title": "Figure Exceeds $1B in Series C Funding at $39B Post-Money Valuation",
        "type": "blog",
        "publisher": "Figure AI",
        "date": "2025-09-16",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Significant investment from Brookfield Asset Management, NVIDIA, Macquarie Capital, Intel Capital, Align Ventures, Tamarack Global, LG Technology Ventures, Salesforce, T-Mobile Ventures and others."
     }
    ],
    "funding_note": "Earlier rounds (Series A in 2023 and Series B in 2024) are not listed because we did not open sources for them in this pass.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.figure.ai/news",
      "title": "Figure news index",
      "type": "site",
      "publisher": "Figure AI",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.figure.ai/news/project-go-big",
      "title": "Project Go-Big",
      "type": "blog",
      "publisher": "Figure AI",
      "date": "2025-09-18",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.figure.ai/news/helix-02",
      "title": "Helix 02",
      "type": "blog",
      "publisher": "Figure AI",
      "date": "2026-01-27",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.figure.ai/news/introducing-index",
      "title": "Introducing Index: Building The World's Largest and Most Diverse Physical Dataset",
      "type": "blog",
      "publisher": "Figure AI",
      "date": "2026-08-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.figure.ai/news/helix-2-5-zero-shot-30-home-generalization",
      "title": "Helix 2.5: zero-shot generalization in 30 homes",
      "type": "blog",
      "publisher": "Figure AI",
      "date": "2026-09-17",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "world-labs",
    "name": "World Labs",
    "type": "start-up",
    "region": "north-america",
    "hq": "San Francisco, United States (not stated on the company's about page)",
    "parent": "",
    "models": [
     {
      "name": "World generation preview (\"Generating Worlds\")",
      "date": "2024-12-02",
      "family": "3d-world",
      "access": "research-only",
      "url": "https://www.worldlabs.ai/blog/generating-worlds",
      "level": "verified",
      "note": "Blog post showing early persistent, navigable 3D worlds explorable in a browser."
     },
     {
      "name": "RTFM (Real-Time Frame Model)",
      "date": "2025-10-16",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://www.worldlabs.ai/blog/rtfm",
      "level": "verified",
      "note": "Generates video in real time as the user interacts; the company calls it a generative world model. Released as a research preview with a browser demo. Second family: 3d-world (explores generated 3D worlds)."
     },
     {
      "name": "Marble",
      "date": "2025-11-12",
      "family": "3d-world",
      "access": "product",
      "url": "https://www.worldlabs.ai/blog/marble-world-model",
      "level": "verified",
      "note": "Generally available web product; worlds export as Gaussian splats, meshes or videos. Developer access through the World API from 2026-01-21."
     },
     {
      "name": "Atlas",
      "date": "2026-09-01",
      "family": "3d-world",
      "access": "research-only",
      "url": "https://www.worldlabs.ai/blog/atlas",
      "level": "verified",
      "note": "Described as an omni world model (multimodal autoregressive diffusion transformer over text, images, video and 3D). Early access with select partners only; the company says it will power future versions of Marble."
     }
    ],
    "technical_position": "World Labs published its first navigable 3D world results on 2024-12-02, released the RTFM real-time frame model as a research preview on 2025-10-16, and made Marble, a product that generates persistent 3D worlds from text, images or video, generally available on 2025-11-12, followed by the World API on 2026-01-21. On 2026-09-01 it put Atlas, a model that generates new views, 3D reconstructions and scene changes over time, into early access with select partners, and on 2026-09-28 it announced a definitive agreement to join AMD.",
    "compute_or_data_signals": "RTFM: the company states that inference runs at interactive frame rates on a single H100 GPU (2025-10-16). Atlas: the company states it was pretrained from scratch on a large multimodal corpus but gives no data, parameter or compute figures (2026-09-01).",
    "funding": [
     {
      "date": "2024-09",
      "round": "Two rounds announced together (names not stated)",
      "amount": "$230 million",
      "valuation": "over $1 billion (reported)",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2024/09/13/fei-fei-lis-world-labs-comes-out-of-stealth-with-230m-in-funding/",
        "title": "Fei-Fei Li's World Labs comes out of stealth with $230M in funding",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2024-09-13",
        "accessed": "2026-10-10"
       }
      ],
      "note": "TechCrunch says the money came over two rounds a couple of months apart, from backers including Andreessen Horowitz, NEA and Radical Ventures; it names no lead. The company's about page lists investors but no amounts."
     },
     {
      "date": "2026-02",
      "round": "New funding (round name not stated)",
      "amount": "$1 billion",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/blog/funding-2026",
        "title": "World Labs Announces New Funding",
        "type": "blog",
        "publisher": "World Labs",
        "date": "2026-02-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://adsknews.autodesk.com/en/news/autodesk-invests-in-world-labs/",
        "title": "Autodesk invests $200 million in World Labs, secures strategic advisor role",
        "type": "blog",
        "publisher": "Autodesk",
        "date": "2026-02-18",
        "accessed": "2026-10-10"
       }
      ],
      "note": "The company names AMD, Autodesk, Emerson Collective, Fidelity Management & Research Company, NVIDIA and Sea among investors and does not name a lead or a valuation. Autodesk's own post says its share was a $200 million strategic investment with an advisor role."
     },
     {
      "date": "2026-09",
      "round": "Acquisition agreement (to join AMD)",
      "amount": "not disclosed",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/blog/amd-announcement",
        "title": "World Labs is Joining AMD",
        "type": "blog",
        "publisher": "World Labs",
        "date": "2026-09-28",
        "accessed": "2026-10-10"
       }
      ],
      "note": "World Labs says it signed a definitive agreement to join AMD, expected to close by the end of 2026 subject to regulatory approvals. Fei-Fei Li is to join AMD as Executive Vice President and Chief Scientist. No price was published."
     }
    ],
    "funding_note": "World Labs has not published a valuation for any round. The price of the AMD transaction was not disclosed.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.worldlabs.ai/blog",
      "title": "World Labs blog and news index",
      "type": "blog",
      "publisher": "World Labs",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/generating-worlds",
      "title": "Generating Worlds",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2024-12-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/rtfm",
      "title": "RTFM: A Real-Time Frame Model",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2025-10-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/marble-world-model",
      "title": "Marble: A Multimodal World Model",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2025-11-12",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/announcing-the-world-api",
      "title": "Announcing the World API",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2026-01-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/atlas",
      "title": "Atlas: A World Model for Spatial Intelligence",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2026-09-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/about",
      "title": "About World Labs",
      "type": "site",
      "publisher": "World Labs",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.worldlabs.ai/blog/amd-announcement",
      "title": "World Labs is Joining AMD",
      "type": "blog",
      "publisher": "World Labs",
      "date": "2026-09-28",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "ami-labs",
    "name": "AMI Labs (Advanced Machine Intelligence)",
    "type": "start-up",
    "region": "europe",
    "hq": "Paris, France (the company lists Paris, New York, Montreal and Singapore)",
    "parent": "",
    "models": [],
    "technical_position": "AMI Labs launched publicly on 2026-03-10 and says it is developing world models that learn abstract representations of sensor data and make predictions in representation space, including action-conditioned world models for planning. We found no released model, paper or code from AMI Labs on its site as of 2026-10-10.",
    "compute_or_data_signals": "",
    "funding": [
     {
      "date": "2026-03",
      "round": "Seed (as reported; the company says \"a round\")",
      "amount": "$1.03B USD (~€890M)",
      "valuation": "$3.5 billion pre-money (reported)",
      "lead_investors": [
       "Cathay Innovation",
       "Greycroft",
       "Hiro Capital",
       "HV Capital",
       "Bezos Expeditions"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://amilabs.xyz/updates",
        "title": "AMI Labs - Updates: Official launch",
        "type": "blog",
        "publisher": "Advanced Machine Intelligence (AMI Labs)",
        "date": "2026-03-10",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://techcrunch.com/2026/03/09/yann-lecuns-ami-labs-raises-1-03-billion-to-build-world-models/",
        "title": "Yann LeCun's AMI Labs raises $1.03B to build world models",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2026-03-09",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Amount and co-leads are from the company's own update of 2026-03-10. The valuation comes only from news reports (TechCrunch). Other named backers include Toyota Ventures, Temasek, NVIDIA, Sea, Samsung and Bpifrance Digital Venture."
     }
    ],
    "funding_note": "The company has not published a valuation.",
    "level": "verified",
    "sources": [
     {
      "url": "https://amilabs.xyz/",
      "title": "AMI Labs: Real World. Real Intelligence.",
      "type": "site",
      "publisher": "Advanced Machine Intelligence (AMI Labs)",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://amilabs.xyz/updates",
      "title": "AMI Labs - Updates: Official launch",
      "type": "blog",
      "publisher": "Advanced Machine Intelligence (AMI Labs)",
      "date": "2026-03-10",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "general-intuition",
    "name": "General Intuition",
    "type": "start-up",
    "region": "north-america",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "MIRA",
      "date": "2026-07",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://mira-wm.com/blog-post/",
      "level": "inferred",
      "note": "Playable multiplayer world model of Rocket League built by General Intuition and Kyutai with Epic Games; 5B-parameter diffusion transformer plus a 600M-parameter video codec; 20 fps for four players. Public browser demo. Training and inference code are open (Apache-2.0, GitHub repo created 2026-07-05) and the dataset is on Hugging Face (kyutai/rocket-science); we found no official model weights. Date inferred from the repo creation date; SiliconANGLE says it was released in June, other reports say 2026-07-07."
     }
    ],
    "technical_position": "General Intuition was spun out of the game-clip platform Medal in 2025-10 and trains agent and world models on gameplay video; TechCrunch reported on 2026-06-25 that it sells the agent model and uses its world model internally as a training environment. In 2026-07 it released MIRA, a real-time multiplayer world model of Rocket League built with Kyutai and Epic Games, with open code and data and a public demo.",
    "compute_or_data_signals": "MIRA: about 10,000 match-hours of bot self-play for training; a released dataset of 4,000 hours (1,000 match-hours across four players) at 720p; 5B-parameter model plus 600M-parameter codec (MIRA project page). TechCrunch reported that most of the 2026-06 round goes to compute through a deal with CoreWeave.",
    "funding": [
     {
      "date": "2025-10",
      "round": "Seed",
      "amount": "$133.7 million (reported; TechCrunch later wrote \"$134 million\")",
      "valuation": "",
      "lead_investors": [
       "Khosla Ventures",
       "General Catalyst"
      ],
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2026/06/25/general-intuitions-2-3b-bet-that-video-games-can-train-ai-agents-for-the-real-world/",
        "title": "General Intuition's $2.3B bet that video games can train AI agents for the real world",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2026-06-25",
        "accessed": "2026-10-10"
       }
      ],
      "note": "We did not find a company or investor announcement page that we could open. Date and leads are from news reports."
     },
     {
      "date": "2026-06",
      "round": "Series A (as reported)",
      "amount": "$320 million",
      "valuation": "$2.3 billion (reported)",
      "lead_investors": [
       "Khosla Ventures"
      ],
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2026/06/25/general-intuitions-2-3b-bet-that-video-games-can-train-ai-agents-for-the-real-world/",
        "title": "General Intuition's $2.3B bet that video games can train AI agents for the real world",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2026-06-25",
        "accessed": "2026-10-10"
       }
      ],
      "note": "TechCrunch: participation from General Catalyst, Jeff Bezos, Eric Schmidt and Nico Rosberg; total disclosed funding $454 million."
     },
     {
      "date": "2026-09",
      "round": "Not named",
      "amount": "$220 million",
      "valuation": "$6.2 billion (reported)",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://siliconangle.com/2026/09/29/world-model-startup-general-intuition-closes-220m-investment/",
        "title": "World model startup General Intuition closes $220M investment",
        "type": "secondary",
        "publisher": "SiliconANGLE",
        "date": "2026-09-29",
        "accessed": "2026-10-10"
       }
      ],
      "note": "SiliconANGLE names Valor Equity Partners, Atreides, Seven Seven Six, Point72, Khosla Ventures and General Catalyst and does not name a lead. The company's homepage says it raised over $650M in the past year, which is consistent with the three reported rounds."
     }
    ],
    "funding_note": "The company's own site states only a total (over $650M in the past year); per-round figures and valuations come from news reports and the company's posts on X, which we could not open.",
    "level": "reported",
    "sources": [
     {
      "url": "https://www.generalintuition.com/",
      "title": "General Intuition",
      "type": "site",
      "publisher": "General Intuition",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://mira-wm.com/blog-post/",
      "title": "MIRA: a playable multiplayer world model",
      "type": "blog",
      "publisher": "General Intuition and Kyutai",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/mira-wm/mira",
      "title": "mira-wm/mira: Code for MIRA: Multiplayer Interactive World Models with Representation Autoencoders",
      "type": "repo",
      "publisher": "General Intuition and Kyutai",
      "date": "2026-07-05",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://techcrunch.com/2026/06/25/general-intuitions-2-3b-bet-that-video-games-can-train-ai-agents-for-the-real-world/",
      "title": "General Intuition's $2.3B bet that video games can train AI agents for the real world",
      "type": "secondary",
      "publisher": "TechCrunch",
      "date": "2026-06-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://siliconangle.com/2026/09/29/world-model-startup-general-intuition-closes-220m-investment/",
      "title": "World model startup General Intuition closes $220M investment",
      "type": "secondary",
      "publisher": "SiliconANGLE",
      "date": "2026-09-29",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "decart",
    "name": "Decart",
    "type": "start-up",
    "region": "multi",
    "hq": "San Francisco, United States, with operations in Israel (TechCrunch 2024-12-19)",
    "parent": "",
    "models": [
     {
      "name": "Oasis",
      "date": "2024-10-31",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://decart.ai/publications/oasis-interactive-ai-video-game-model",
      "level": "verified",
      "note": "Real-time playable Minecraft-like world generated frame by frame from keyboard and mouse input, built with Etched. Code and the weights of a smaller model (Etched/oasis-500m on Hugging Face, MIT licence tag) were released with a live demo of a larger checkpoint. Runs at 360p and 20 fps on NVIDIA H100s according to the post."
     },
     {
      "name": "Lucy 2",
      "date": "2026-01-26",
      "family": "video-generation",
      "access": "api",
      "url": "https://decart.ai/publications/lucy-2-introducing-sota-video-generation-in-realtime",
      "level": "verified",
      "note": "Real-time video transformation model (1080p, 30 fps per the post); Decart describes its Lucy line as a world model for immersive experiences. Lucy 2.5 followed on 2026-07-16. Earlier Lucy Edit open-weight checkpoints exist on Hugging Face (decart-ai)."
     },
     {
      "name": "Oasis 3",
      "date": "2026-06-10",
      "family": "action-video",
      "access": "api",
      "url": "https://decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
      "level": "verified",
      "note": "Generative world model for physical AI, starting with autonomous vehicles; conditioned on robot actions and outputs synchronized multi-camera views; available through an API from launch."
     }
    ],
    "technical_position": "Decart released Oasis, a real-time playable world model built with Etched, with open weights for a 500M-parameter version on 2024-10-31, and has since released real-time video transformation models (Lucy 2 on 2026-01-26, Lucy 2.5 on 2026-07-16). On 2026-06-10 it launched Oasis 3, an action-conditioned multi-camera driving world model offered through an API.",
    "compute_or_data_signals": "Oasis 3: runs at 22 FPS at 512x768 with under 200 ms latency on CoreWeave and NVIDIA HGX B200 systems, and was trained using the NVIDIA Physical AI Open Dataset (Decart, 2026-06-10). Oasis: 360p at 20 fps on NVIDIA H100s (Decart, 2024-10-31).",
    "funding": [
     {
      "date": "2024-10",
      "round": "Seed",
      "amount": "$21 million (reported)",
      "valuation": "",
      "lead_investors": [
       "Sequoia Capital"
      ],
      "level": "reported",
      "sources": [
       {
        "url": "https://www.sequoiacap.com/article/partnering-with-decart-the-future-of-ai-generated-experiences/",
        "title": "Partnering with Decart: The Future of AI-Generated Experiences",
        "type": "blog",
        "publisher": "Sequoia Capital",
        "date": "2024-10-31",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://techcrunch.com/2024/12/19/decart-adds-another-32m-at-a-500m-valuation/",
        "title": "Decart adds another $32M at a $500M valuation",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2024-12-19",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Sequoia's own post says it led the seed round but gives no amount; the $21 million figure and Zeev Ventures' participation come from TechCrunch."
     },
     {
      "date": "2024-12",
      "round": "Series A",
      "amount": "$32 million",
      "valuation": "over $500 million post-money (reported)",
      "lead_investors": [
       "Benchmark"
      ],
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2024/12/19/decart-adds-another-32m-at-a-500m-valuation/",
        "title": "Decart adds another $32M at a $500M valuation",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2024-12-19",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "date": "2025-08",
      "round": "Series B",
      "amount": "$100 million",
      "valuation": "$3.1 billion (reported)",
      "lead_investors": [],
      "level": "reported",
      "sources": [
       {
        "url": "https://fortune.com/2025/08/07/exclusive-decart-raises-100-million-at-a-3-1-billion-valuation-chasing-the-future-of-real-time-creative-ai/",
        "title": "Exclusive: Decart raises $100 million at a $3.1 billion valuation",
        "type": "secondary",
        "publisher": "Fortune",
        "date": "2025-08-07",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Fortune names Sequoia Capital, Benchmark and Zeev Ventures as participating existing investors and reports $153 million raised in 11 months."
     },
     {
      "date": "2026-05",
      "round": "Not named by the company",
      "amount": "$300 million",
      "valuation": "nearly $4 billion (reported)",
      "lead_investors": [
       "Radical Ventures"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://decart.ai/publications/decart-raises-300m-tech-leaders-back-the-company-as-both-customers-and-investors",
        "title": "Decart Raises $300M: Tech Leaders Back the Company as Both Customers and Investors",
        "type": "blog",
        "publisher": "Decart",
        "date": "2026-05-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://pulse2.com/decart-300-million-raised-at-nearly-4-billion-valuation-raised-to-make-switching-ai-chips-easier/",
        "title": "Decart: $300 Million Raised At Nearly $4 Billion Valuation",
        "type": "secondary",
        "publisher": "Pulse 2.0",
        "date": "2026-05-19",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Company post names NVIDIA, Atreides Management, Valor Equity, Adobe Ventures, Toyota Ventures and eBay Ventures as participants, with returning investors including Sequoia Capital and Benchmark. The valuation and the 'more than $450 million to date' total come from Pulse 2.0; some outlets call this round a Series B, which would conflict with Fortune's use of Series B for the 2025-08 round."
     }
    ],
    "funding_note": "Decart has not published valuations on its own site; all valuation figures are from news reports.",
    "level": "verified",
    "sources": [
     {
      "url": "https://decart.ai/blog",
      "title": "Decart AI Lab | Resources (publications and news index)",
      "type": "blog",
      "publisher": "Decart",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://decart.ai/publications/oasis-interactive-ai-video-game-model",
      "title": "Oasis: A Universe in a Transformer",
      "type": "blog",
      "publisher": "Decart",
      "date": "2024-10-31",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/Etched/oasis-500m",
      "title": "Etched/oasis-500m model card",
      "type": "repo",
      "publisher": "Etched (with Decart)",
      "date": "2024-10-31",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://decart.ai/publications/lucy-2-introducing-sota-video-generation-in-realtime",
      "title": "Introducing Lucy 2: SOTA realtime world transformation model",
      "type": "blog",
      "publisher": "Decart",
      "date": "2026-01-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
      "title": "Introducing Oasis 3: First Interactive World Model for Physical AI",
      "type": "blog",
      "publisher": "Decart",
      "date": "2026-06-10",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "odyssey",
    "name": "Odyssey",
    "type": "start-up",
    "region": "north-america",
    "hq": "Palo Alto, United States (reported by news; not stated on the company's about page)",
    "parent": "",
    "models": [
     {
      "name": "Explorer",
      "date": "2024-12-18",
      "family": "3d-world",
      "access": "research-only",
      "url": "https://odyssey.systems/introducing-explorer",
      "level": "verified",
      "note": "Image-to-world model that turns an image into a 3D world output as Gaussian splats; trialled with Garden Studios in London."
     },
     {
      "name": "Odyssey-1",
      "date": "2025-05-28",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://odyssey.systems/introducing-odyssey-1",
      "level": "verified",
      "note": "Real-time playable world model released as a research preview, streaming up to 30 FPS from H100 clusters."
     },
     {
      "name": "Odyssey-2",
      "date": "2025-10-27",
      "family": "interactive-world",
      "access": "api",
      "url": "https://odyssey.systems/introducing-odyssey-2",
      "level": "verified",
      "note": "General-purpose interactive video world model; an Odyssey-2 Pro version is named in the company's 2026-02-12 post; the site now offers API access."
     },
     {
      "name": "Odyssey-2 Max",
      "date": "2026-04-21",
      "family": "video-generation",
      "access": "research-only",
      "url": "https://odyssey.systems/introducing-odyssey-2-max",
      "level": "verified",
      "note": "Largest Odyssey-2 model, presented for physical accuracy of world simulation."
     },
     {
      "name": "Starchild-1",
      "date": "2026-05-17",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://odyssey.systems/introducing-starchild-1",
      "level": "verified",
      "note": "Real-time world model that generates both visuals and sound. The post index lists it on 2026-05-18."
     },
     {
      "name": "Agora-2",
      "date": "2026-09-21",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://odyssey.systems/introducing-agora-2",
      "level": "verified",
      "note": "Multi-agent world model for up to 20 humans and agents in one shared simulation, released as a playable research preview; follows Agora-1 (2026-05-18). The post index lists it on 2026-09-24."
     },
     {
      "name": "Odyssey-3",
      "date": "2026-09-15",
      "family": "action-video",
      "access": "research-only",
      "url": "https://odyssey.systems/meet-odyssey-3",
      "level": "verified",
      "note": "Foundation world model (autoregressive diffusion transformer) that Odyssey adapts to control robot arms, a humanoid (with Flexion) and a car; launched as a research preview with developer API access on 2026-10-08 after being introduced on 2026-09-15. Second family: video-generation."
     }
    ],
    "technical_position": "Odyssey moved from a 3D world generator (Explorer, 2024-12-18) to real-time playable video world models (Odyssey-1 on 2025-05-28, Odyssey-2 on 2025-10-27) and in 2026 added a scaled model (Odyssey-2 Max), a real-time audio-visual model (Starchild-1) and multi-agent world models (Agora-1 and Agora-2). On 2026-09-15 and 2026-10-08 it presented Odyssey-3, a foundation world model it adapted to control robot arms, a humanoid and a car with tens of hours or less of task data.",
    "compute_or_data_signals": "Odyssey-1 research preview streamed at up to 30 FPS from H100 GPU clusters in the US and EU, at a stated infrastructure cost of $1-$2 per user-hour (2025-05-28). Odyssey-3: driving policy trained on 20 hours of driving data with the backbone frozen; robot arm control from tens of hours of demonstrations (2026-10-08). The Series B post names AWS as preferred cloud and work with Amazon's Annapurna Labs on Trainium chips.",
    "funding": [
     {
      "date": "2024-12",
      "round": "Investment by Ed Catmull (board member)",
      "amount": "not disclosed",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/introducing-explorer",
        "title": "World Models for Film, Gaming, and Beyond",
        "type": "blog",
        "publisher": "Odyssey",
        "date": "2024-12-18",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Odyssey announced that Pixar co-founder Ed Catmull joined its board and invested."
     },
     {
      "date": "2026-02",
      "round": "Strategic investment",
      "amount": "not disclosed",
      "valuation": "",
      "lead_investors": [],
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/investment-from-nvidia-and-samsung",
        "title": "Odyssey Announces Investment from NVentures and Samsung Next",
        "type": "blog",
        "publisher": "Odyssey",
        "date": "2026-02-12",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Investment from NVentures (NVIDIA's venture arm) and Samsung Next."
     },
     {
      "date": "2026-06",
      "round": "Series B",
      "amount": "$310 million",
      "valuation": "$1.45 billion",
      "lead_investors": [
       "Natural Capital"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/our-series-b",
        "title": "Our $310 Million Fundraise to Accelerate World Simulation",
        "type": "blog",
        "publisher": "Odyssey",
        "date": "2026-06-17",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://techcrunch.com/2026/06/17/world-model-maker-odyssey-nabs-1-45b-valuation-backed-by-amazon-and-other-big-names/",
        "title": "World model maker Odyssey nabs $1.45B valuation backed by Amazon and other big names",
        "type": "secondary",
        "publisher": "TechCrunch",
        "date": "2026-06-17",
        "accessed": "2026-10-10"
       }
      ],
      "note": "Participants named by the company: Amazon, GV, AMD Ventures, EQT, IQT and others. TechCrunch reports $337 million raised to date."
     }
    ],
    "funding_note": "We found no primary or news source giving the amounts of rounds before the 2026 Series B; TechCrunch's $337 million total implies about $27 million earlier, which we have not confirmed round by round.",
    "level": "verified",
    "sources": [
     {
      "url": "https://odyssey.systems/writing",
      "title": "The latest from Odyssey (post index)",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-explorer",
      "title": "World Models for Film, Gaming, and Beyond",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2024-12-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-odyssey-1",
      "title": "Introducing Odyssey-1: A Playable World Model",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2025-05-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-odyssey-2",
      "title": "Introducing Odyssey-2: A General-Purpose World Model",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2025-10-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-odyssey-2-max",
      "title": "Introducing Odyssey-2 Max: Scaled World Simulation",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2026-04-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-starchild-1",
      "title": "Starchild-1: The First Real-Time Multimodal World Model",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2026-05-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-agora-2",
      "title": "Introducing Agora-2: Advancing Multi-Agent World Simulation",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2026-09-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/introducing-odyssey-3",
      "title": "Introducing Odyssey-3: A General-Purpose Physical Intelligence",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2026-09-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://odyssey.systems/meet-odyssey-3",
      "title": "Meet Odyssey-3: Our Most Powerful Foundation World Model",
      "type": "blog",
      "publisher": "Odyssey",
      "date": "2026-10-08",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "runway",
    "name": "Runway",
    "type": "start-up",
    "region": "north-america",
    "hq": "New York, United States (Crunchbase News)",
    "parent": "",
    "models": [
     {
      "name": "Gen-4.5",
      "date": "2025-12-01",
      "family": "video-generation",
      "access": "product",
      "url": "https://runway.com/research/introducing-runway-gen-4.5",
      "level": "verified",
      "note": "Text- and image-to-video model available in Runway's app and API; GWM-1 is built on top of it. Runway described video generators such as Gen-2 as early forms of general world models on 2023-12-11."
     },
     {
      "name": "GWM-1 (GWM Worlds, GWM Robotics, GWM Avatars)",
      "date": "2025-12-11",
      "family": "interactive-world",
      "access": "product",
      "url": "https://runway.com/research/introducing-runway-gwm-1",
      "level": "verified",
      "note": "Autoregressive model built on Gen-4.5 that generates frame by frame in real time and is controlled by camera pose, robot commands or audio; three post-trained variants. GWM Robotics is action-conditioned (second family: action-video). Access differs by variant; we did not confirm general availability of each."
     },
     {
      "name": "Solaris",
      "date": "2026-08-31",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://runway.com/news/research/introducing-solaris",
      "level": "verified",
      "note": "Interface world model that generates a user interface frame by frame in response to mouse actions."
     },
     {
      "name": "GWM Worlds 2",
      "date": "2026-09-03",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://runway.com/research/introducing-gwm-worlds-2",
      "level": "verified",
      "note": "Research preview generating continuous 720p video at 24 fps with 48,000 Hz audio, steered by text actions and camera motion."
     }
    ],
    "technical_position": "Runway announced a research programme on general world models on 2023-12-11 and on 2025-12-11 released GWM-1, a real-time action-controllable model family for explorable worlds, robotics and conversational avatars built on its Gen-4.5 video model. In 2026 it showed GWM Worlds 2 with generated audio (2026-09-03), an interface world model called Solaris (2026-08-31), and Praxis-1, an open-weight robot action model built on its video pretraining that it is testing with early partners (2026-09).",
    "compute_or_data_signals": "GWM Worlds 2 generates 720p video at 24 fps and 48,000 Hz audio in real time (2026-09-03). Gen-4.5 page states the model was built on NVIDIA Hopper and Blackwell GPUs (2025-12-01). The Praxis-1 post states that simulating robot policies inside Runway's world model predicts real-world results with 0.95 correlation (2026-09); the post does not give the number of policies or tasks.",
    "funding": [
     {
      "date": "2025-04",
      "round": "Series D",
      "amount": "over $300M (company); $308 million (reported)",
      "valuation": "$3.3 billion (reported)",
      "lead_investors": [
       "General Atlantic"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/news/runway-series-d-funding",
        "title": "Towards a new media ecosystem with world simulators",
        "type": "blog",
        "publisher": "Runway",
        "date": "2025-04-03",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://news.crunchbase.com/venture/gen-ai-video-startup-unicorn-runway-seriese-raise/",
        "title": "Crunchbase News on Runway's Series E",
        "type": "secondary",
        "publisher": "Crunchbase News",
        "date": "2026-02-10",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Company post names Fidelity Management & Research Company, Baillie Gifford, NVIDIA and SoftBank Vision Fund 2 as participants. The exact $308 million and the valuation come from Crunchbase News."
     },
     {
      "date": "2026-02",
      "round": "Series E",
      "amount": "$315 million",
      "valuation": "$5.3 billion (reported)",
      "lead_investors": [
       "General Atlantic"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/news/runway-series-e-funding",
        "title": "New Funding to Scale World Simulation",
        "type": "blog",
        "publisher": "Runway",
        "date": "2026-02-10",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://news.crunchbase.com/venture/gen-ai-video-startup-unicorn-runway-seriese-raise/",
        "title": "Crunchbase News on Runway's Series E",
        "type": "secondary",
        "publisher": "Crunchbase News",
        "date": "2026-02-10",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Company post names NVIDIA, Adobe Ventures, AllianceBernstein, AMD Ventures, Fidelity Management & Research Company, Mirae Asset, Emphatic Capital, Felicis and Premji Invest. Crunchbase News reports $860 million raised since 2018."
     }
    ],
    "funding_note": "Runway's own posts state amounts and leads but not valuations. Rounds before 2025 (Runway was founded in 2018) are not listed here because they predate its world-model work; Crunchbase News gives $860 million raised in total.",
    "level": "verified",
    "sources": [
     {
      "url": "https://runway.com/research",
      "title": "Runway Research",
      "type": "site",
      "publisher": "Runway",
      "date": "",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/introducing-general-world-models",
      "title": "Introducing General World Models",
      "type": "blog",
      "publisher": "Runway",
      "date": "2023-12-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/introducing-runway-gen-4.5",
      "title": "Runway Gen-4.5: State-of-the-Art AI Video Generation",
      "type": "blog",
      "publisher": "Runway",
      "date": "2025-12-01",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/introducing-runway-gwm-1",
      "title": "Introducing GWM-1",
      "type": "blog",
      "publisher": "Runway",
      "date": "2025-12-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/news/research/introducing-solaris",
      "title": "Introducing Solaris",
      "type": "blog",
      "publisher": "Runway",
      "date": "2026-08-31",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/introducing-gwm-worlds-2",
      "title": "Introducing GWM Worlds 2",
      "type": "blog",
      "publisher": "Runway",
      "date": "2026-09-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/introducing-praxis-1",
      "title": "Introducing Praxis-1",
      "type": "blog",
      "publisher": "Runway",
      "date": "2026-09",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "luma-ai",
    "name": "Luma AI",
    "type": "start-up",
    "region": "north-america",
    "hq": "Palo Alto, United States (dateline of the company's Ray3 release)",
    "parent": "",
    "models": [
     {
      "name": "Ray3",
      "date": "2025-09-18",
      "family": "video-generation",
      "access": "product",
      "url": "https://lumalabs.ai/news/ray3",
      "level": "verified",
      "note": "Video model the company calls a reasoning video model, with 16-bit HDR output; available in the Dream Machine platform and in Adobe Firefly."
     },
     {
      "name": "Ray3.2",
      "date": "2026-06-09",
      "family": "video-generation",
      "access": "api",
      "url": "https://lumalabs.ai/news/introducing-ray-3-2",
      "level": "verified",
      "note": "Update to Ray3 with frame-level control; the company says the full Ray control surface is available as an API for the first time."
     }
    ],
    "technical_position": "Luma AI makes video generation models (Ray3 on 2025-09-18, Ray3.2 on 2026-06-09) sold through its Dream Machine product and API. The company says it is training large world models and, in its Series C post of 2025-11-19, said it would scale multimodal models to simulate the physical world; we found no product it names as a world model as of 2026-10-11.",
    "compute_or_data_signals": "Luma says it is partnering with HUMAIN on Project Halo, a 2GW compute supercluster that begins deploying in Q1 2026 and finishes by 2028-29, for training and inference (2025-11-19).",
    "funding": [
     {
      "date": "2025-11",
      "round": "Series C",
      "amount": "900M",
      "valuation": "around $4 billion (reported)",
      "lead_investors": [
       "HUMAIN"
      ],
      "level": "verified",
      "sources": [
       {
        "url": "https://lumalabs.ai/news/series-c",
        "title": "AGI is multimodal and reality is the dataset of AGI",
        "type": "blog",
        "publisher": "Luma AI",
        "date": "2025-11-19",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://cairoscene.com/Business/Saudi-Based-Humain-Leads-900M-Series-C-Round-for-Luma-AI",
        "title": "Saudi-Based Humain Leads $900M Series C Round for Luma AI",
        "type": "secondary",
        "publisher": "CairoScene",
        "date": "2025-11-20",
        "accessed": "2026-10-11"
       }
      ],
      "note": "Company post: led by Humain with significant participation from AMD, and existing investors Andreessen Horowitz, Omniva, Amplify Partners and Matrix Partners. The valuation is from news reports only."
     }
    ],
    "funding_note": "Luma has not published a valuation. Rounds before 2025 are not listed because we did not open primary or news sources for them.",
    "level": "verified",
    "sources": [
     {
      "url": "https://lumalabs.ai/news/ray3",
      "title": "Luma AI launches Ray3",
      "type": "press-release",
      "publisher": "Luma AI",
      "date": "2025-09-18",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://lumalabs.ai/news/introducing-ray-3-2",
      "title": "Luma Introduces Ray3.2 Model & API",
      "type": "blog",
      "publisher": "Luma AI",
      "date": "2026-06-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://lumalabs.ai/news/luma-joins-forces-with-humain",
      "title": "Luma is Partnering with HUMAIN to Accelerate the Arrival of Multimodal AGI",
      "type": "blog",
      "publisher": "Luma AI",
      "date": "2025-05-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://lumalabs.ai/news/series-c",
      "title": "AGI is multimodal and reality is the dataset of AGI",
      "type": "blog",
      "publisher": "Luma AI",
      "date": "2025-11-19",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "dynamics-lab",
    "name": "Dynamics Lab",
    "type": "start-up",
    "region": "",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Mirage (later renamed Magica 1)",
      "date": "2025-07",
      "family": "interactive-world",
      "access": "research-only",
      "url": "http://web.archive.org/web/20250702231845/https://blog.dynamicslab.ai/",
      "level": "verified",
      "note": "Research preview of a real-time, text- and controller-driven game world generator with playable browser demos (GTA-style and Forza-style). The official blog survives only as an archived copy; by 2025-12 the product was renamed Magica."
     },
     {
      "name": "Mirage 2 (later renamed Magica 2)",
      "date": "2025-08",
      "family": "interactive-world",
      "access": "research-only",
      "url": "https://the-decoder.com/mirage-2-allows-users-to-turn-sketches-and-photos-into-interactive-game-worlds/",
      "level": "reported",
      "note": "Generates playable worlds from uploaded images or sketches with text commands during play; announced on the company's X account (not opened) and reported by The Decoder on 2025-08-22."
     }
    ],
    "technical_position": "Dynamics Lab released Mirage, a real-time generative game engine with playable browser demos, as a research preview in 2025-07 and a second version in 2025-08 that turns uploaded images into playable worlds; by 2025-12 its blog called the product Magica. As of 2026-10-10 the company domain dynamicslab.ai shows a parked page and blog.dynamicslab.ai does not resolve, so we could not confirm current activity.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "We found no announced or reported funding round for Dynamics Lab. We looked at the archived company blog, news coverage of Mirage and Mirage 2, and general web search.",
    "level": "reported",
    "sources": [
     {
      "url": "http://web.archive.org/web/20250702231845/https://blog.dynamicslab.ai/",
      "title": "Mirage: AI UGC game engine (archived copy of the official blog, 2025-07-02)",
      "type": "blog",
      "publisher": "Dynamics Lab",
      "date": "2025-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "http://web.archive.org/web/20251202203502/https://blog.dynamicslab.ai/",
      "title": "Magica: AI UGC game engine (archived copy of the official blog, 2025-12-02)",
      "type": "blog",
      "publisher": "Dynamics Lab",
      "date": "2025-12",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://the-decoder.com/mirage-2-allows-users-to-turn-sketches-and-photos-into-interactive-game-worlds/",
      "title": "Mirage 2 allows users to turn sketches and photos into interactive game worlds",
      "type": "secondary",
      "publisher": "The Decoder",
      "date": "2025-08-22",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "id": "tencent",
    "name": "Tencent (Hunyuan)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "HunyuanWorld 1.0",
      "date": "2025-07-26",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-1.0",
      "level": "verified",
      "note": "Generates explorable 3D worlds from text or images via panoramas. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     },
     {
      "name": "Hunyuan-GameCraft 1.0",
      "date": "2025-08-14",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/Hunyuan-GameCraft-1.0",
      "level": "verified",
      "note": "Game video generation controlled by keyboard and mouse actions. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     },
     {
      "name": "HunyuanWorld-Voyager",
      "date": "2025-09-02",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-Voyager",
      "level": "verified",
      "note": "RGB-D video generation conditioned on camera paths for 3D-consistent exploration. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     },
     {
      "name": "HunyuanWorld-Mirror (HunyuanWorld 1.1)",
      "date": "2025-10-22",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-Mirror",
      "level": "verified",
      "note": "Feed-forward 3D reconstruction from video or multi-view images. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     },
     {
      "name": "HY-World 1.5 (WorldPlay)",
      "date": "2025-12-17",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/HY-WorldPlay",
      "level": "verified",
      "note": "Streaming video diffusion model for real-time interactive worlds with geometric consistency; 8B (HunyuanVideo-based) and 5B (Wan-based) variants. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     },
     {
      "name": "HY-World 2.0",
      "date": "2026-04-16",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/Tencent-Hunyuan/HY-World-2.0",
      "level": "verified",
      "note": "Multi-modal model that outputs 3D world representations (meshes, Gaussian splats) instead of video; weights released in parts from 2026-04-16 to 2026-05-18. Repository licence is a custom licence that GitHub does not identify (NOASSERTION)."
     }
    ],
    "technical_position": "Tencent's Hunyuan team has released a sequence of open-weight world models: HunyuanWorld 1.0 for 3D world generation (2025-07-26), the action-controlled Hunyuan-GameCraft (2025-08-14), Voyager and WorldMirror (2025-09 and 2025-10), the real-time interactive HY-World 1.5 (2025-12-17) and HY-World 2.0 (2026-04-16), which generates 3D scenes directly. It also maintains the open HunyuanVideo generators that some of these models build on.",
    "compute_or_data_signals": "HY-World 1.5: 8B (HunyuanVideo-based) and 5B (Wan-based) model variants (README). No training compute figures found on the repository pages.",
    "funding": [],
    "funding_note": "Tencent does not disclose spending on world-model research.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-1.0",
      "title": "Tencent-Hunyuan/HunyuanWorld-1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2025-07-26",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/Hunyuan-GameCraft-1.0",
      "title": "Tencent-Hunyuan/Hunyuan-GameCraft-1.0: High-dynamic Interactive Game Video Generation",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2025-08-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-Voyager",
      "title": "Tencent-Hunyuan/HunyuanWorld-Voyager: Interactive RGBD video generation conditioned on camera input",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2025-09-02",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-Mirror",
      "title": "Tencent-Hunyuan/HunyuanWorld-Mirror: WorldMirror: universal 3D reconstruction",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2025-10-22",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/HY-WorldPlay",
      "title": "Tencent-Hunyuan/HY-WorldPlay: HY-World 1.5: interactive world modeling with real-time latency",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2025-12-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Tencent-Hunyuan/HY-World-2.0",
      "title": "Tencent-Hunyuan/HY-World-2.0: HY-World 2.0: A Multi-Modal World Model for Reconstructing, Generating, and Simulating 3D Worlds",
      "type": "repo",
      "publisher": "Tencent Hunyuan",
      "date": "2026-04-16",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "alibaba",
    "name": "Alibaba (Tongyi Wan, DAMO Academy)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "Alibaba Group",
    "models": [
     {
      "name": "Wan 2.1",
      "date": "2025-02-25",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/Wan-Video/Wan2.1",
      "level": "verified",
      "note": "Apache-2.0 open video generation models. Alibaba does not call Wan a world model on the repository page, but several robot world models (Unitree, DAMO, Tsinghua Vidar) are built on Wan backbones."
     },
     {
      "name": "Wan 2.2",
      "date": "2025-07-28",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/Wan-Video/Wan2.2",
      "level": "verified",
      "note": "Apache-2.0; includes a 5B text-image-to-video model (TI2V-5B)."
     },
     {
      "name": "WorldVLA / RynnVLA-002",
      "date": "2025-06-23",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/alibaba-damo-academy/RynnVLA-002",
      "level": "verified",
      "note": "Autoregressive model that generates both the next image given an action and actions given images; upgraded to RynnVLA-002 on 2025-11-10; Apache-2.0."
     },
     {
      "name": "RynnWorld-4D",
      "date": "2026-07-07",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/alibaba-damo-academy/RynnWorld-4D",
      "level": "verified",
      "note": "4D embodied world model co-generating RGB, depth and optical flow for manipulation; built on Wan2.2-TI2V-5B."
     },
     {
      "name": "RynnWorld-Teleop",
      "date": "2026-07-02",
      "family": "action-video",
      "access": "research-only",
      "url": "https://github.com/alibaba-damo-academy/RynnWorld-Teleop",
      "level": "inferred",
      "note": "Action-conditioned world model for digital teleoperation; date is the repository creation date."
     }
    ],
    "technical_position": "Alibaba's Tongyi lab released the open-weight Wan video models (Wan 2.1 on 2025-02-25, Wan 2.2 on 2025-07-28), which other groups use as backbones for robot world models. Its DAMO Academy released WorldVLA (2025-06-23, later RynnVLA-002), which predicts both images and robot actions, and the RynnWorld robot world models in 2026-07.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Alibaba does not disclose spending on world-model research. Later Wan versions offered only through APIs were not checked.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/Wan-Video/Wan2.1",
      "title": "Wan-Video/Wan2.1: Wan: Open and Advanced Large-Scale Video Generative Models",
      "type": "repo",
      "publisher": "Alibaba Tongyi Wan",
      "date": "2025-02-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Wan-Video/Wan2.2",
      "title": "Wan-Video/Wan2.2: Wan: Open and Advanced Large-Scale Video Generative Models",
      "type": "repo",
      "publisher": "Alibaba Tongyi Wan",
      "date": "2025-07-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/alibaba-damo-academy/RynnVLA-002",
      "title": "alibaba-damo-academy/RynnVLA-002: RynnVLA-002 (formerly WorldVLA): A Unified Vision-Language-Action and World Model",
      "type": "repo",
      "publisher": "Alibaba DAMO Academy",
      "date": "2025-06-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/alibaba-damo-academy/RynnWorld-4D",
      "title": "alibaba-damo-academy/RynnWorld-4D: RynnWorld-4D: 4D Embodied World Models for Robotic Manipulation",
      "type": "repo",
      "publisher": "Alibaba DAMO Academy",
      "date": "2026-07-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/alibaba-damo-academy/RynnWorld-Teleop",
      "title": "alibaba-damo-academy/RynnWorld-Teleop: RynnWorld-Teleop: An Action-Conditioned World Model for Digital Teleoperation",
      "type": "repo",
      "publisher": "Alibaba DAMO Academy",
      "date": "2026-07-02",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "bytedance",
    "name": "ByteDance (Seed)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "ByteDance",
    "models": [
     {
      "name": "VideoWorld",
      "date": "2025-01-15",
      "family": "latent-dynamics",
      "access": "open-weights",
      "url": "https://github.com/ByteDance-Seed/VideoWorld",
      "level": "inferred",
      "note": "Generative model that learns from unlabeled video; proposes a latent dynamics model separating action dynamics from appearance. Date is the repository creation date; Apache-2.0 per GitHub metadata."
     },
     {
      "name": "Seedance 1.0",
      "date": "2025-06",
      "family": "video-generation",
      "access": "api",
      "url": "https://seed.bytedance.com/en/seedance",
      "level": "inferred",
      "note": "Multi-shot text- and image-to-video model offered by API; the page cites leaderboards as of 2025-06-09, so the month is inferred. The Seed site lists Seedance 2.5 as its current video model; we did not open a page dating it. ByteDance does not call Seedance a world model on this page."
     }
    ],
    "technical_position": "ByteDance Seed released VideoWorld (repository 2025-01-15, CVPR 2025), which learns world knowledge from unlabeled video, and builds the Seedance video generators, with Seedance 2.5 listed as current on its site in 2026-10. Its robot models (GR series) are policies and are not listed here.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "ByteDance does not disclose spending on world-model research.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/ByteDance-Seed/VideoWorld",
      "title": "ByteDance-Seed/VideoWorld: VideoWorld: learning world models from unlabeled videos (CVPR 2025)",
      "type": "repo",
      "publisher": "ByteDance Seed",
      "date": "2025-01-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://seed.bytedance.com/en/seedance",
      "title": "Seedance (ByteDance Seed model page)",
      "type": "site",
      "publisher": "ByteDance Seed",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "kuaishou",
    "name": "Kuaishou (Kling)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [],
    "technical_position": "Kuaishou builds the Kling video generation models. We could not open a primary source listing Kling versions or any Kuaishou world model: the Kling release-notes page did not render its content and Kuaishou's investor site returned HTTP 403.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Kuaishou does not disclose spending on world-model research.",
    "level": "unknown",
    "sources": [
     {
      "url": "https://kling.ai/release-note/release-notes",
      "title": "Kling AI Release Notes (content did not load)",
      "type": "site",
      "publisher": "Kuaishou",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "ant-robbyant",
    "name": "Ant Group (Robbyant / LingBot)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "Ant Group",
    "models": [
     {
      "name": "LingBot-World",
      "date": "2026-01-29",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/Robbyant/lingbot-world",
      "level": "verified",
      "note": "Open world simulator built from video generation; real-time interaction with under 1 second latency at 16 frames per second; Apache-2.0."
     },
     {
      "name": "LingBot-VA",
      "date": "2026-01-29",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/Robbyant/lingbot-va",
      "level": "verified",
      "note": "Autoregressive video-action world model that predicts video and robot actions in one sequence; Apache-2.0; RSS 2026."
     },
     {
      "name": "LingBot-World-Infinity (lingbot-world-v2)",
      "date": "2026-07-09",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/Robbyant/lingbot-world-v2",
      "level": "verified",
      "note": "14B and 1.3B variants; a distilled real-time variant drives 720p video at 60 fps; CC BY-NC-SA 4.0 (non-commercial)."
     }
    ],
    "technical_position": "Ant Group's embodied-AI unit Robbyant released LingBot-World, an open real-time interactive world model, and LingBot-VA, a model that predicts video and robot actions together, on 2026-01-29. It followed with LingBot-World-Infinity on 2026-07-09, with 14B and 1.3B open-weight variants and a real-time 720p mode.",
    "compute_or_data_signals": "LingBot-World-Infinity: 14B and 1.3B parameter models (README).",
    "funding": [],
    "funding_note": "We found no separate outside funding for Robbyant on its repository pages; Ant Group does not disclose spending on it.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/Robbyant/lingbot-world",
      "title": "Robbyant/lingbot-world: LingBot-World: Advancing Open-source World Models",
      "type": "repo",
      "publisher": "Robbyant (Ant Group)",
      "date": "2026-01-29",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Robbyant/lingbot-va",
      "title": "Robbyant/lingbot-va: LingBot-VA: Causal World Modeling for Robot Control",
      "type": "repo",
      "publisher": "Robbyant (Ant Group)",
      "date": "2026-01-29",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/Robbyant/lingbot-world-v2",
      "title": "Robbyant/lingbot-world-v2: LingBot-World-Infinity: Infinite Worlds with Versatile Interactions",
      "type": "repo",
      "publisher": "Robbyant (Ant Group)",
      "date": "2026-07-09",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "sensetime",
    "name": "SenseTime",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "OpenDWM",
      "date": "2025-01-15",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/SenseTime-FVG/OpenDWM",
      "level": "inferred",
      "note": "Open driving world models that generate multi-view driving video or LiDAR from text and road layouts; MIT; experimental interactive generation with CARLA (2025-03-17). Date is the repository creation date."
     }
    ],
    "technical_position": "SenseTime Research released OpenDWM (2025-01), an open code base of driving world models that generate multi-view driving video and LiDAR from text and road layouts. We could not open SenseTime's own announcements of its 'Kaiwu' world model, so that product is not listed.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "SenseTime does not disclose spending on world-model research.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/SenseTime-FVG/OpenDWM",
      "title": "SenseTime-FVG/OpenDWM: Open Driving World Models (OpenDWM)",
      "type": "repo",
      "publisher": "SenseTime Research (with Southeast University)",
      "date": "2025-01-15",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "shanghai-ai-lab",
    "name": "Shanghai AI Laboratory",
    "type": "lab",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Aether",
      "date": "2025-03-28",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/InternRobotics/Aether",
      "level": "verified",
      "note": "Geometry-aware unified world model (4D reconstruction, video prediction, planning); MIT licence; ICCV 2025. Published under the InternRobotics GitHub organisation; author affiliations not checked on the paper."
     },
     {
      "name": "Vista (OpenDriveLab)",
      "date": "2024-05-28",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/OpenDriveLab/Vista",
      "level": "inferred",
      "note": "Driving world model, NeurIPS 2024, Apache-2.0. OpenDriveLab is linked to Shanghai AI Laboratory and the University of Hong Kong; we did not confirm the lead affiliation on the paper."
     },
     {
      "name": "InternW0-Δ",
      "date": "2026-09-24",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/InternRobotics/InternW0-Delta",
      "level": "inferred",
      "note": "World action model trained with 20K+ hours of open data; code MIT. Date is the repository creation date; weights availability not checked."
     }
    ],
    "technical_position": "Groups at Shanghai AI Laboratory released Aether (2025-03-28), an open geometry-aware world model, and in 2026-09 InternW0-Δ, a world action model trained on more than 20,000 hours of open data. The related OpenDriveLab group released the Vista driving world model in 2024-05.",
    "compute_or_data_signals": "InternW0-Δ: 20K+ hours of open data (repository title).",
    "funding": [],
    "funding_note": "Shanghai AI Laboratory is a state-backed lab and does not publish a budget for world-model research.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/InternRobotics/Aether",
      "title": "InternRobotics/Aether: Aether: Geometric-Aware Unified World Modeling",
      "type": "repo",
      "publisher": "Shanghai AI Laboratory (InternRobotics)",
      "date": "2025-03-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/InternRobotics/InternW0-Delta",
      "title": "InternRobotics/InternW0-Delta: InternW0-Δ: A World Action Model with 20K+ Hours of Open Data",
      "type": "repo",
      "publisher": "Shanghai AI Laboratory (InternRobotics)",
      "date": "2026-09-24",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/OpenDriveLab/Vista",
      "title": "OpenDriveLab/Vista: Vista: A Generalizable Driving World Model",
      "type": "repo",
      "publisher": "OpenDriveLab",
      "date": "2024-05-28",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "baai",
    "name": "Beijing Academy of Artificial Intelligence (BAAI)",
    "type": "lab",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Emu3",
      "date": "2024-09-26",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/baaivision/Emu3",
      "level": "inferred",
      "note": "Next-token prediction model over text, images and video; Apache-2.0; date is the repository creation date."
     },
     {
      "name": "Emu3.5",
      "date": "2025-10-29",
      "family": "video-generation",
      "access": "open-weights",
      "url": "https://github.com/baaivision/Emu3.5",
      "level": "inferred",
      "note": "BAAI describes it as a native multimodal 'world learner' that predicts the next state across vision and language; Apache-2.0; web and mobile apps live from 2025-11-28. The family is our closest fit (it generates interleaved images and text)."
     }
    ],
    "technical_position": "BAAI released Emu3 (2024-09) and Emu3.5 (2025-10), open multimodal models trained by next-token prediction, and describes Emu3.5 as a world learner that predicts the next state across vision and language. Emu3.5 also became available as web and mobile apps on 2025-11-28.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "BAAI is a non-profit research institute and does not publish a budget for world-model work.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/baaivision/Emu3",
      "title": "baaivision/Emu3: Emu3: Next-Token Prediction is All You Need",
      "type": "repo",
      "publisher": "BAAI",
      "date": "2024-09-26",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/baaivision/Emu3.5",
      "title": "baaivision/Emu3.5: Emu3.5: Native Multimodal Models are World Learners",
      "type": "repo",
      "publisher": "BAAI",
      "date": "2025-10-29",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "skywork",
    "name": "Skywork AI (Kunlun Tech)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Matrix-Game 1.0",
      "date": "2025-05-12",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "level": "verified",
      "note": "MIT licence."
     },
     {
      "name": "Matrix-Game 2.0",
      "date": "2025-08-12",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "level": "verified",
      "note": "Interactive world foundation model for real-time long video generation; MIT."
     },
     {
      "name": "Matrix-3D",
      "date": "2025-08-11",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/SkyworkAI/Matrix-3D",
      "level": "inferred",
      "note": "Generates explorable 3D scenes from panorama video; MIT; date is the repository creation date."
     },
     {
      "name": "Matrix-Game 3.0",
      "date": "2026-03-27",
      "family": "interactive-world",
      "access": "open-weights",
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "level": "verified",
      "note": "Real-time streaming interactive world model with long-horizon memory; MIT."
     }
    ],
    "technical_position": "Skywork AI has released the open Matrix-Game interactive world models under the MIT licence (1.0 on 2025-05-12, 2.0 on 2025-08-12, 3.0 on 2026-03-27). It also released Matrix-3D for explorable 3D scenes in 2025-08.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "We did not check Skywork AI's corporate structure or funding; it is commonly described as part of Kunlun Tech, which we did not confirm at a source.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/SkyworkAI/Matrix-Game",
      "title": "SkyworkAI/Matrix-Game: Matrix-Game series of open-source world models",
      "type": "repo",
      "publisher": "Skywork AI",
      "date": "2025-05-12",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/SkyworkAI/Matrix-3D",
      "title": "SkyworkAI/Matrix-3D: Matrix-3D: explorable 3D scenes from panorama videos",
      "type": "repo",
      "publisher": "Skywork AI",
      "date": "2025-08-11",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "manycore",
    "name": "Manycore Tech (SpatialVerse)",
    "type": "big-tech",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "SpatialGen",
      "date": "2025-08-24",
      "family": "3d-world",
      "access": "open-weights",
      "url": "https://github.com/manycore-research/SpatialGen",
      "level": "inferred",
      "note": "Layout-guided 3D indoor scene generation; MIT; 3DV 2026; date is the repository creation date. Its SpatialLM (2025-03) is a 3D scene understanding model, not a world model."
     }
    ],
    "technical_position": "Manycore Tech's research group released SpatialGen (2025-08), an open model that generates 3D indoor scenes from layouts, alongside SpatialLM (2025-03) for 3D scene understanding. We did not check its funding or listing status.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/manycore-research/SpatialGen",
      "title": "manycore-research/SpatialGen: SpatialGen: Layout-guided 3D Indoor Scene Generation",
      "type": "repo",
      "publisher": "Manycore Tech",
      "date": "2025-08-24",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "agibot",
    "name": "AgiBot (智元机器人)",
    "type": "robot-company",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "EnerVerse-AC (EVAC)",
      "date": "2025-05-14",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/AgibotTech/EnerVerse-AC",
      "level": "inferred",
      "note": "Action-conditioned video world model; released weights trained only on the open AgiBot World dataset; CC BY-NC-SA 4.0; baseline for the AgiBot World Challenge world-model track. Date is the repository creation date."
     },
     {
      "name": "Genie Envisioner (GE-Base, GE-Act, GE-Sim)",
      "date": "2025-08-08",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/AgibotTech/Genie-Envisioner-V1",
      "level": "verified",
      "note": "World foundation platform: GE-Base video world model, GE-Act action decoder and GE-Sim simulator; technical report 2025-08-08, weights 2025-08-14; GE-Act v2.0 released 2026-09-10."
     },
     {
      "name": "GE-Sim 2.0",
      "date": "2026-05-28",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/AgibotTech/GE-Sim-V2",
      "level": "verified",
      "note": "Action-conditioned multi-view video world simulator driven by 16-D joint actions for closed-loop policy evaluation; weights released 2026-06-25; Apache-2.0 / CC BY-NC-SA 4.0."
     }
    ],
    "technical_position": "AgiBot has released a series of open robot world models: EnerVerse-AC (2025-05), the Genie Envisioner platform that pairs a video world model with an action decoder (2025-08-08), and GE-Sim 2.0, a closed-loop simulator for evaluating manipulation policies (report 2026-05-28, weights 2026-06-25). It also runs world-model tracks in the AgiBot World Challenge.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/AgibotTech/EnerVerse-AC",
      "title": "AgibotTech/EnerVerse-AC: EnerVerse-AC: Envisioning Embodied Environments with Action Condition",
      "type": "repo",
      "publisher": "AgiBot",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/AgibotTech/Genie-Envisioner-V1",
      "title": "AgibotTech/Genie-Envisioner-V1: Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
      "type": "repo",
      "publisher": "AgiBot",
      "date": "2025-08-08",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/AgibotTech/GE-Sim-V2",
      "title": "AgibotTech/GE-Sim-V2: GE-Sim 2.0: closed-loop video world simulator for robotic manipulation",
      "type": "repo",
      "publisher": "AgiBot",
      "date": "2026-05-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.agibot.com.cn/news",
      "title": "AgiBot news page (动态速递)",
      "type": "site",
      "publisher": "AgiBot",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "galbot",
    "name": "Galbot (银河通用)",
    "type": "robot-company",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "WAM-TTT",
      "date": "2026-07",
      "family": "world-action",
      "access": "research-only",
      "url": "http://www.galbot.com/news/",
      "level": "reported",
      "note": "Known only from a news item title on Galbot's own news page (source: Tencent Tech, 2026-07-16) describing a world action model with test-time training that adapts to new kitchens without human action labels. We did not open the article or a paper."
     }
    ],
    "technical_position": "Galbot builds humanoid robots and robot foundation models; its own news page lists a 2026-07-16 report on WAM-TTT, a world action model that adapts at test time. We could not open a primary technical description of it.",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds.",
    "level": "reported",
    "sources": [
     {
      "url": "http://www.galbot.com/news/",
      "title": "Galbot news page (lists 'WAM-TTT' item dated 2026-07-16)",
      "type": "site",
      "publisher": "Galbot",
      "date": "",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "unitree",
    "name": "Unitree Robotics (宇树科技)",
    "type": "robot-company",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "UnifoLM-WMA-0",
      "date": "2025-09-15",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/unitreerobotics/unifolm-world-model-action",
      "level": "verified",
      "note": "Open world-model-action architecture: a video world model predicts future frames and drives action generation, and can also run as an interactive simulator; trained first on Open X-Embodiment data then on Unitree G1 data."
     },
     {
      "name": "UnifoLM-WLA-1.0",
      "date": "2026-09-28",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/unitreerobotics/unifolm-wla",
      "level": "inferred",
      "note": "6B-parameter humanoid foundation model built on 'interaction-centric world modeling' (README); base weights released 2026-09-28; Apache-2.0 code. Family is our reading; the README describes an action expert, not explicit frame prediction."
     }
    ],
    "technical_position": "Unitree released UnifoLM-WMA-0 on 2025-09-15, an open framework in which a video world model predicts future frames to guide robot actions, and on 2026-09-28 released UnifoLM-WLA-1.0, a 6B-parameter humanoid foundation model that it says is built on interaction-centric world modeling.",
    "compute_or_data_signals": "UnifoLM-WLA-1.0: 6B parameters (README).",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds. Reports of a stock-market listing were not checked.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/unitreerobotics/unifolm-world-model-action",
      "title": "unitreerobotics/unifolm-world-model-action: UnifoLM-WMA-0: A World-Model-Action Framework",
      "type": "repo",
      "publisher": "Unitree Robotics",
      "date": "2025-09-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/unitreerobotics/unifolm-wla",
      "title": "unitreerobotics/unifolm-wla: UnifoLM-WLA-1.0",
      "type": "repo",
      "publisher": "Unitree Robotics",
      "date": "2026-09-28",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "gigaai",
    "name": "GigaAI (极佳科技)",
    "type": "start-up",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "GigaWorld-0",
      "date": "2025-11-25",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/open-gigaai/giga-world-0",
      "level": "inferred",
      "note": "World model used as a data engine for embodied AI; 2B video models; Apache-2.0; date is the repository creation date."
     },
     {
      "name": "GigaWorld-Policy",
      "date": "2026-03-03",
      "family": "world-action",
      "access": "open-weights",
      "url": "https://github.com/open-gigaai/giga-world-policy",
      "level": "inferred",
      "note": "World action model jointly modeling actions and future observations, with 85 ms latency for local deployment; date is the repository creation date."
     },
     {
      "name": "GigaWorld-1",
      "date": "2026-07",
      "family": "action-video",
      "access": "open-weights",
      "url": "https://github.com/open-gigaai/giga-world-1",
      "level": "verified",
      "note": "World model for robot policy evaluation; technical report and partial weights released 2026-07; Apache-2.0."
     }
    ],
    "technical_position": "GigaAI released GigaWorld-0 (2025-11), an open world model used to generate training data for robots, GigaWorld-Policy (2026-03), a world action model, and GigaWorld-1 (2026-07), a world model for evaluating robot policies. It also hosted the world-model track of a CVPR 2026 challenge.",
    "compute_or_data_signals": "GigaWorld-0: 2B-parameter video models (README).",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://github.com/open-gigaai/giga-world-0",
      "title": "open-gigaai/giga-world-0: GigaWorld-0: World Models as Data Engine to Empower Embodied AI",
      "type": "repo",
      "publisher": "GigaAI",
      "date": "2025-11-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/open-gigaai/giga-world-policy",
      "title": "open-gigaai/giga-world-policy: GigaWorld-Policy: An Efficient Action-Centered World–Action Model",
      "type": "repo",
      "publisher": "GigaAI",
      "date": "2026-03-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/open-gigaai/giga-world-1",
      "title": "open-gigaai/giga-world-1: GigaWorld-1: A Roadmap to World Models for Robot Policy Evaluation",
      "type": "repo",
      "publisher": "GigaAI",
      "date": "2026-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "id": "shengshu",
    "name": "Shengshu Technology (生数科技, Vidu)",
    "type": "start-up",
    "region": "china",
    "hq": "",
    "parent": "",
    "models": [
     {
      "name": "Vidar",
      "date": "2025-07",
      "family": "world-action",
      "access": "research-only",
      "url": "https://github.com/thu-ml/vidar",
      "level": "inferred",
      "note": "Video foundation model for robots (arXiv 2507.12898); repository is under Tsinghua's thu-ml organisation; Shengshu's role is from the lead and was not confirmed on the paper."
     },
     {
      "name": "Motubrain",
      "date": "2026-04",
      "family": "world-action",
      "access": "research-only",
      "url": "https://github.com/shengshu-ai/Motubrain",
      "level": "verified",
      "note": "Unified world action model jointly modeling video dynamics and actions (arXiv 2604.27792)."
     },
     {
      "name": "Vidu S1 / S2",
      "date": "2026-07",
      "family": "interactive-world",
      "access": "product",
      "url": "https://github.com/shengshu-ai/Vidu-S",
      "level": "verified",
      "note": "Real-time interactive video generation; S1 available 2026-07, S2 available to try at vidu.com/vidu-stream from 2026-09."
     },
     {
      "name": "Motus2",
      "date": "2026-09",
      "family": "world-action",
      "access": "research-only",
      "url": "https://github.com/shengshu-ai/Motus2",
      "level": "verified",
      "note": "General world model for dexterous manipulation (arXiv 2608.30237); code and checkpoints planned for release during 2026-09."
     }
    ],
    "technical_position": "Shengshu, maker of the Vidu video generator, has moved into robot world models: Vidar (2025-07, with Tsinghua), the world action models Motubrain (2026-04) and Motus2 (2026-09), and the real-time interactive Vidu S models (2026-07 and 2026-09). It also published minWM, a tutorial framework for real-time interactive world models (2026-05).",
    "compute_or_data_signals": "",
    "funding": [],
    "funding_note": "Funding rounds were not checked: the shared web-search budget ran out before this organisation was reached, and its own news pages we opened did not list rounds.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/thu-ml/vidar",
      "title": "thu-ml/vidar: Vidar and Vidarc: video foundation model for robotics",
      "type": "repo",
      "publisher": "Tsinghua University TSAIL (thu-ml), with Shengshu",
      "date": "2025-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/shengshu-ai/Motubrain",
      "title": "shengshu-ai/Motubrain: Motubrain: An Advanced World Action Model for Robot Control",
      "type": "repo",
      "publisher": "Shengshu Technology",
      "date": "2026-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/shengshu-ai/Motus2",
      "title": "shengshu-ai/Motus2: Motus2: A Self-Evolving General World Model for Dexterous Manipulation",
      "type": "repo",
      "publisher": "Shengshu Technology",
      "date": "2026-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/shengshu-ai/Vidu-S",
      "title": "shengshu-ai/Vidu-S: Vidu S: Real-Time Interactive, Editable, and Spatial Video Generation",
      "type": "repo",
      "publisher": "Shengshu Technology",
      "date": "2026-07",
      "accessed": "2026-10-11"
     }
    ]
   }
  ],
  "notes": "Big-tech spending on world models is not disclosed by any of the companies covered (Alphabet/Google DeepMind and Waymo's world model, OpenAI, Meta, Microsoft, NVIDIA, Tesla, Tencent, Alibaba, ByteDance, Kuaishou, Ant Group, SenseTime). Their funding lists are empty by design; only figures they published (data hours, GPU counts) are recorded. Valuations: most start-ups do not state valuations in their own posts. Exceptions with a company-stated valuation: Odyssey ($1.45 billion, 2026-06), Skild AI ($1.5 billion in 2024-07; over $14 billion in 2026-01), Figure ($39 billion post-money, 2025-09) and Wayve ($8.6 billion post-money, 2026-02). All other valuations in this file are from news reports and are marked as such. Rounds reported only by news (no company announcement found): Physical Intelligence (all rounds), General Intuition (all rounds; the company site states only 'over $650M in the past year'), Decart before 2026-05, Skild AI Series B, World Labs 2024, Runway and Luma valuations. Reported talks that we did not treat as closed rounds: Physical Intelligence about $1 billion at more than $11 billion (Bloomberg, 2026-03-27); 1X up to $1 billion at about $10 billion and SoftBank discussing a controlling stake (news, 2025-2026); World Labs at about $5 billion before its 2026-02 round (news). Deals: World Labs signed a definitive agreement on 2026-09-28 to join AMD (price not disclosed, closing expected by end of 2026). OpenAI shut down the Sora 2 API on 2026-09-24 after notice on 2026-03-24, and the Sora app reportedly closed on 2026-04-26. Chinese start-up funding (AgiBot, Galbot, Unitree, GigaAI, Shengshu, Manycore) was not checked: the shared web-search budget for this session ran out before these organisations were reached, and their own news pages we opened did not list rounds. These funding lists are empty because of that gap, not because there were no rounds. Not covered because of the search limit: Chinese car makers' driving world models (NIO, XPeng, Li Auto, Huawei), Kuaishou's Kling versions, SenseTime's Kaiwu world model, the Beijing Humanoid Robot Innovation Center (X-Humanoid), Covariant, Helm.ai, comma.ai, SpAItial, Moonlake AI and Overworld. Tesla's world simulator could not be confirmed at a primary source (tesla.com returned HTTP 403 and its SEC filings do not mention it). Blocked primary sites: openai.com, help.openai.com, tesla.com, ir.tesla.com, microsoft.com research pages (one opened through WebFetch), pi.website (HTTP 429; an archived copy dated 2026-10-03 was used), dynamicslab.ai (now a parked page; archived copies of its blog were used). The new family 'world-action' (models that predict future frames and output robot actions) is used for 1XWM as a policy, WorldVLA/RynnVLA-002, LingBot-VA, Genie Envisioner, UnifoLM-WMA-0 and UnifoLM-WLA-1.0, GigaWorld-Policy, InternW0-Δ, Vidar, Motubrain, Motus2 and Galbot's WAM-TTT. Policies built on world-model pretraining but described without frame prediction (Runway Praxis-1, Odyssey-3 driving and robot adaptations) are not given this family. Taxonomy friction: Waabi World and the Waymo World Model are learned driving and sensor simulators; 'learned-simulator' in the brief refers to physics engines, so we used it for Waabi World and action-video for Waymo. Copilot4D predicts LiDAR point clouds rather than video. Emu3.5 is a multimodal next-token model that BAAI calls a world learner; no family fits exactly. Name collisions: Meta's 2026 'Muse Video' (Meta Superintelligence Labs video generator) is unrelated to Microsoft's 2025 'Muse' (WHAM). 'Genie' (Google DeepMind world models) is unrelated to AgiBot's 'Genie Envisioner' and 'Genie Sim'."
 },
 "usecases": {
  "use_cases": [
   {
    "id": "robot-training-data",
    "name": "Robot training data generation",
    "plain": "The world model makes new robot training videos or trajectories, either by changing the look of existing recordings and simulator output (lighting, objects, backgrounds) or by imagining a robot doing new tasks. Robot companies want this because recording real robot demonstrations takes people, hardware and time for every task and setting.",
    "maturity": "product",
    "maturity_basis": "NVIDIA offers its Cosmos world models for this purpose under an open model license and names robot makers that use them (Agility Robotics, 1X, Skild AI, Agile Robots and others), and Runway sells a robotics data augmentation service; the size of the gains is still measured mostly by the model makers themselves.",
    "evidence": [
     {
      "org": "Agility Robotics; NVIDIA",
      "what": "NVIDIA named Agility Robotics as an early adopter of Cosmos Transfer and Omniverse for large-scale synthetic data to train its robot models, and Agility's chief technology officer said Cosmos lets it scale photorealistic training data beyond what it can collect in the real world.",
      "date": "2025-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-announces-major-release-of-cosmos-world-foundation-models-and-physical-ai-data-tools",
        "title": "NVIDIA Announces Major Release of Cosmos World Foundation Models and Physical AI Data Tools",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-world-foundation-model-platform-to-accelerate-physical-ai-development",
        "title": "NVIDIA Launches Cosmos World Foundation Model Platform to Accelerate Physical AI Development",
        "type": "press-release",
        "date": "2025-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Agile Robots; NVIDIA",
      "what": "NVIDIA says Agile Robots is using Cosmos 3 to generate action-conditioned robot data for its policy development, to create varied task trajectories at scale.",
      "date": "2026-05",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://blogs.nvidia.com/blog/cosmos-3-physical-ai-open-world-foundation-model/",
        "title": "How Cosmos 3 Helps Physical AI Think Before It Acts",
        "type": "blog",
        "date": "2026-05-31",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Skild AI; FieldAI; Generalist AI; NVIDIA",
      "what": "NVIDIA says Skild AI and FieldAI are building general robot brains using Cosmos world models for data generation, and that Generalist AI is using Cosmos to explore generating synthetic data.",
      "date": "2026-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-and-global-robotics-leaders-take-physical-ai-to-the-real-world",
        "title": "NVIDIA and Global Robotics Leaders Take Physical AI to the Real World",
        "type": "press-release",
        "date": "2026-03-16",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Lightwheel; Moon Surgical; Skild AI; NVIDIA",
      "what": "NVIDIA lists Lightwheel, Moon Surgical and Skild AI as using Cosmos Transfer to speed up robot training by simulating varied conditions at scale.",
      "date": "2025-08",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-opens-portals-to-world-of-robotics-with-new-omniverse-libraries-cosmos-physical-ai-models-and-ai-computing-infrastructure",
        "title": "NVIDIA Opens Portals to World of Robotics With New Omniverse Libraries, Cosmos Physical AI Models and AI Computing Infrastructure",
        "type": "press-release",
        "date": "2025-08-11",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "1X Technologies; NVIDIA",
      "what": "NVIDIA says 1X is using Cosmos Predict and Cosmos Transfer to train its NEO Gamma humanoid robot; in January 2025 NVIDIA said 1X used the Cosmos Tokenizer for its 1X World Model Challenge dataset.",
      "date": "2025-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-announces-major-release-of-cosmos-world-foundation-models-and-physical-ai-data-tools",
        "title": "NVIDIA Announces Major Release of Cosmos World Foundation Models and Physical AI Data Tools",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-world-foundation-model-platform-to-accelerate-physical-ai-development",
        "title": "NVIDIA Launches Cosmos World Foundation Model Platform to Accelerate Physical AI Development",
        "type": "press-release",
        "date": "2025-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA",
      "what": "NVIDIA released Cosmos world foundation models in January 2025 under an open model license for generating synthetic training data for robots and vehicles, and in May 2026 released Cosmos 3, which it says can also generate robot actions; NVIDIA lists Doosan Robotics, LG Electronics, Samsung Electronics and Skild AI among robotics developers building on Cosmos.",
      "date": "2026-05",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-3-the-open-frontier-foundation-model-for-physical-ai",
        "title": "NVIDIA Launches Cosmos 3, the Open Frontier Foundation Model for Physical AI",
        "type": "press-release",
        "date": "2026-05-31",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-world-foundation-model-platform-to-accelerate-physical-ai-development",
        "title": "NVIDIA Launches Cosmos World Foundation Model Platform to Accelerate Physical AI Development",
        "type": "press-release",
        "date": "2025-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway offers a robotics platform built on its GWM-1 world model that turns existing robot trajectories into new environments, lighting conditions and object layouts; access is by request.",
      "date": "2025-12",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://runway.com/product/robotics",
        "title": "Runway Robotics: Build smarter robots with general world models",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA Research",
      "what": "NVIDIA says its researchers used the GR00T-Dreams blueprint, which generates synthetic robot motion data with Cosmos, to develop the GR00T N1.5 robot model in 36 hours instead of the nearly three months that manual human data collection would have taken (internal use).",
      "date": "2025-05",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-powers-humanoid-robot-industry-with-cloud-to-robot-computing-platforms-for-physical-ai",
        "title": "NVIDIA Powers Humanoid Robot Industry With Cloud-to-Robot Computing Platforms for Physical AI",
        "type": "press-release",
        "date": "2025-05-18",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA (GR00T N1 report)",
      "what": "NVIDIA fine-tuned video generation models on 88 hours of its own teleoperation data and generated 827 hours of video, then reports average gains of +4.2%, +8.8% and +6.8% on the RoboCasa simulation benchmark (30, 100 and 300 demonstrations per task) and +5.8% across 8 real tasks on a GR-1 humanoid when co-training with these generated trajectories.",
      "date": "2025-03",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2503.14734",
        "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots",
        "type": "paper",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://arxiv.org/html/2503.14734v2",
        "title": "GR00T N1 (HTML v2, sections 2.2, 3.2 and 4.4)",
        "type": "paper",
        "date": "2025-03-27",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA (DreamGen)",
      "what": "The DreamGen paper reports that a humanoid robot learned 22 new behaviors in seen and unseen environments from video-world-model data, while teleoperation data came from only a single pick-and-place task in one environment.",
      "date": "2025-05",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2505.12705",
        "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
        "type": "paper",
        "date": "2025-05-19",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "GigaAI",
      "what": "GigaAI's GigaWorld-0 paper describes a world model built as a data engine for robot policies and reports that its GigaBrain-0 model, trained on the generated data, improved task success on physical robots without real-world interaction during training.",
      "date": "2025-11",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2511.19861",
        "title": "GigaWorld-0: World Models as Data Engine to Empower Embodied AI",
        "type": "paper",
        "date": "2025-11-25",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Generating the data takes a lot of computing: NVIDIA reports that its 827 hours of generated robot video took about 105,000 L40 GPU hours (1.5 days on 3,600 L40 GPUs).",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/html/2503.14734v2",
        "title": "GR00T N1 (HTML v2, section 3.2)",
        "type": "paper",
        "date": "2025-03-27",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Video models output pictures without the robot's actions, so the actions have to be estimated afterwards with a separate model, which adds another source of error.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2505.12705",
        "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
        "type": "paper",
        "date": "2025-05-19",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://www.1x.tech/discover/world-model-self-learning",
        "title": "1X World Model | From Video to Action: A New Way Robots Learn",
        "type": "blog",
        "date": "2026-01-12",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Generated videos can look correct while breaking physical rules such as object consistency, depth and contact, and training only on expert demonstrations made AgiBot's simulator show a towel moving with a gripper that never grasped it.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/world-model-self-learning",
        "title": "1X World Model | From Video to Action: A New Way Robots Learn",
        "type": "blog",
        "date": "2026-01-12",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://ge-sim-v2.github.io/",
        "title": "GE-Sim2 | Genie Envisioner World Simulator 2.0",
        "type": "site",
        "date": "2026-04",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "The reported gains come from the model makers' own experiments and are a few percentage points in NVIDIA's GR00T N1 tests; we found no independent study that measured them.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://arxiv.org/html/2503.14734v2",
        "title": "GR00T N1 (HTML v2, section 4.4)",
        "type": "paper",
        "date": "2025-03-27",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "NVIDIA said in August 2025 that its Cosmos world foundation models had been downloaded over 2 million times.",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-opens-portals-to-world-of-robotics-with-new-omniverse-libraries-cosmos-physical-ai-models-and-ai-computing-infrastructure",
        "title": "NVIDIA Opens Portals to World of Robotics With New Omniverse Libraries, Cosmos Physical AI Models and AI Computing Infrastructure",
        "type": "press-release",
        "date": "2025-08-11",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "what": "On 2026-10-10 the Hugging Face API reported all-time download counts of 811,479 for nvidia/Cosmos-Predict2.5-2B, 504,027 for nvidia/Cosmos-Transfer2.5-2B and 939,169 for nvidia/Cosmos3-Nano; Hugging Face counts file downloads, not users.",
      "level": "verified",
      "sources": [
       {
        "url": "https://huggingface.co/api/models?author=nvidia&search=Cosmos&sort=downloads&direction=-1&limit=40&expand[]=downloads&expand[]=downloadsAllTime&expand[]=createdAt&expand[]=likes",
        "title": "Hugging Face Hub API: NVIDIA Cosmos models with download counts",
        "type": "index",
        "date": "2026-10-10",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "what": "NVIDIA launched a Cosmos Coalition in May 2026 with Agile Robots, Black Forest Labs, Generalist, LTX, Runway and Skild AI, and in July 2026 said ten Japanese companies including FANUC, Hitachi, Sony Group and Yaskawa Electric intend to join.",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-3-the-open-frontier-foundation-model-for-physical-ai",
        "title": "NVIDIA Launches Cosmos 3, the Open Frontier Foundation Model for Physical AI",
        "type": "press-release",
        "date": "2026-05-31",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/japans-robotics-and-manufacturing-leaders-build-on-nvidia-cosmos-to-advance-physical-ai-frontier",
        "title": "Japan's Robotics and Manufacturing Leaders Build on NVIDIA Cosmos to Advance Physical AI Frontier",
        "type": "press-release",
        "date": "2026-07-15",
        "accessed": "2026-10-10"
       }
      ]
     }
    ]
   },
   {
    "id": "robot-policy-evaluation",
    "name": "Robot policy evaluation",
    "plain": "The world model stands in for the real world: a robot's control software is run inside the model's predicted video and scored, so teams can compare versions before testing on real robots. This matters because real-robot testing is slow and costly, and a test lab cannot cover every home or factory a robot will meet.",
    "maturity": "pilot",
    "maturity_basis": "1X uses its world model internally to choose between robot model versions, and Runway and NVIDIA offer world-model policy evaluation, but the published checks against real robots cover eight policies or fewer per study and were run by the developers themselves.",
    "evidence": [
     {
      "org": "Runway",
      "what": "Runway ran eight robot policies from the RoboArena benchmark inside its GWM-Robotics world model (1,450 simulated rollouts, over 16,000 human ratings) and reports a Pearson correlation of 0.95 between simulated and real-world scores; it sells this as offline policy evaluation and says it builds GWM-Robotics with partners including NVIDIA and Berkshire Grey.",
      "date": "2026-02",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
        "title": "Accelerating Robot Policy Evaluation with General World Models",
        "type": "blog",
        "date": "2026-02-27",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://runway.com/product/robotics",
        "title": "Runway Robotics: Build smarter robots with general world models",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA",
      "what": "NVIDIA describes its open Cosmos Predict 2.5 and Cosmos Transfer 2.5 world models as enabling robot policy evaluation in simulation, alongside synthetic data generation.",
      "date": "2026-01",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-releases-new-physical-ai-models-as-global-partners-unveil-next-generation-robots",
        "title": "NVIDIA Releases New Physical AI Models as Global Partners Unveil Next-Generation Robots",
        "type": "press-release",
        "date": "2026-01-05",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "1X Technologies",
      "what": "1X uses its 1X World Model internally to compare robot policies on the same starting scenes, pick the best training checkpoint and re-test models on hard cases; it reports that predicted success rates correlated with real-world task scores, and that with a 15% real success gap a world model with 70% accuracy picks the better policy 90% of the time.",
      "date": "2025-06",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/redwood-ai-world-model",
        "title": "1X World Model",
        "type": "blog",
        "date": "2025-06-16",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Google DeepMind (Gemini Robotics Team)",
      "what": "Google DeepMind fine-tuned its Veo video model to simulate a two-armed robot and checked its predictions against more than 1,600 real-world trials of eight Gemini Robotics policy versions on five tasks; for changed scenes it reports a rank error (MMRV) of 0.06 and a Pearson correlation of 0.86 between predicted and real success rates.",
      "date": "2025-12",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2512.10675",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
        "type": "paper",
        "date": "2025-12-11",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://arxiv.org/html/2512.10675v2",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator (HTML v2, section 4)",
        "type": "paper",
        "date": "2026-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "AgiBot",
      "what": "AgiBot released Genie Envisioner World Simulator 2.0 (GE-Sim 2.0), an action-conditioned video simulator with built-in task scoring for evaluating and training robot policies, and says it ranked first on the WorldArena Track 1 leaderboard in May 2026.",
      "date": "2026-04",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://agibot.com/article/231/detail/57.html",
        "title": "AGIBOT Unveils Genie Envisioner 2.0, Advancing World Models into Scalable World Simulators for Embodied AI",
        "type": "press-release",
        "date": "2026-04-10",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://ge-sim-v2.github.io/",
        "title": "GE-Sim2 | Genie Envisioner World Simulator 2.0",
        "type": "site",
        "date": "2026-04",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://agibot.com/article/231/detail/71.html",
        "title": "AGIBOT's Genie Envisioner-Sim 2.0 Ranks No. 1 on WorldArena Benchmark",
        "type": "press-release",
        "date": "2026-05-29",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "GigaAI (GigaWorld Team)",
      "what": "GigaAI compared 7 video world models as policy evaluators using over 324,000 simulated policy rollouts paired with real robot runs, and reports that long, action-faithful consistency matters more than short-term visual realism for a good evaluator.",
      "date": "2026-07",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2607.02642",
        "title": "GigaWorld-1: A Roadmap to Build World Models for Robot Policy Evaluation",
        "type": "paper",
        "date": "2026-07-02",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Stanford University; Tsinghua University (Ctrl-World)",
      "what": "Ctrl-World, trained on the DROID dataset (95k trajectories, 564 scenes), ranks robot policies without real-robot rollouts, and its authors report that fine-tuning on successful trajectories imagined in the model raised policy success by 44.7%.",
      "date": "2025-10",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2510.10125",
        "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
        "type": "paper",
        "date": "2025-10-11",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://ctrl-world.github.io",
        "title": "Ctrl-World project page",
        "type": "site",
        "date": "2025-10",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Stanford University (WorldGym)",
      "what": "WorldGym runs robot policies inside an action-conditioned video model with a vision-language model as judge, and its authors report that success rates in the model highly correlate with real-world success rates and keep the ranking of policy versions.",
      "date": "2025-05",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2506.00613",
        "title": "WorldGym: World Model as An Environment for Policy Evaluation",
        "type": "paper",
        "date": "2025-05-31",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Midea Group; East China Normal University (WorldEval)",
      "what": "WorldEval turns a video model into a policy evaluator and its authors report a strong correlation with real-world policy performance and use it to flag dangerous actions by new robot models.",
      "date": "2025-05",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2505.19017",
        "title": "WorldEval: World Model as Real-World Robot Policies Evaluator",
        "type": "paper",
        "date": "2025-05-25",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://worldeval.github.io",
        "title": "WorldEval project page",
        "type": "site",
        "date": "2025-05",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Contact-rich interactions with small objects are still hard to simulate, and Google DeepMind shows a case where an object appears from nowhere during a grasp.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/html/2512.10675v2",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator (HTML v2, section 7)",
        "type": "paper",
        "date": "2026-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Rollouts are short: Google DeepMind's study used 8-second episodes and Runway reports up to 30 seconds, while many household tasks take minutes.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/html/2512.10675v2",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator (HTML v2, section 7)",
        "type": "paper",
        "date": "2026-01-06",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
        "title": "Accelerating Robot Policy Evaluation with General World Models",
        "type": "blog",
        "date": "2026-02-27",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Scoring still needs people: both Google DeepMind and Runway had humans grade the generated videos, and DeepMind found that predicted success rates were lower than real ones, so the models are used to rank policies rather than to predict exact success rates.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/html/2512.10675v2",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator (HTML v2)",
        "type": "paper",
        "date": "2026-01-06",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
        "title": "Accelerating Robot Policy Evaluation with General World Models",
        "type": "blog",
        "date": "2026-02-27",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "1X reports that its world model struggles with objects it has not seen in training, which limits evaluation in new homes.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/redwood-ai-world-model",
        "title": "1X World Model",
        "type": "blog",
        "date": "2025-06-16",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Each published validation compares eight policies or fewer and was run by the model's own developers; we found no independent replication.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
        "title": "Accelerating Robot Policy Evaluation with General World Models",
        "type": "blog",
        "date": "2026-02-27",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://arxiv.org/abs/2512.10675",
        "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator",
        "type": "paper",
        "date": "2025-12-11",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "Runway sells offline policy evaluation as part of its Runway Robotics platform, with access by request and no public price.",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/product/robotics",
        "title": "Runway Robotics: Build smarter robots with general world models",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "what": "AgiBot announced in May 2026 that its world simulator ranked first on the WorldArena Track 1 leaderboard, which shows robot makers competing on world-model benchmarks.",
      "level": "verified",
      "sources": [
       {
        "url": "https://agibot.com/article/231/detail/71.html",
        "title": "AGIBOT's Genie Envisioner-Sim 2.0 Ranks No. 1 on WorldArena Benchmark",
        "type": "press-release",
        "date": "2026-05-29",
        "accessed": "2026-10-10"
       }
      ]
     }
    ]
   },
   {
    "id": "robot-planning-control",
    "name": "Robot planning and control",
    "plain": "The robot uses a world model to imagine what will happen if it takes different actions and then picks the action that leads to the goal, or the world model is built into the software that outputs the robot's movements. The aim is a robot that can handle new objects and tasks by drawing on what the model learned from large amounts of video, with fewer task-specific demonstrations.",
    "maturity": "pilot",
    "maturity_basis": "1X runs a world-model-based policy on its NEO robots in tests and in its own factory, and Runway and NVIDIA offer world models that output robot actions, but we found no robot sold to customers whose published control software is confirmed to be world-model based.",
    "evidence": [
     {
      "org": "1X Technologies",
      "what": "1X integrated its 1X World Model into the NEO humanoid as a robot policy: a 14B-parameter video model, trained further on 900 hours of first-person human video and 70 hours of robot data, imagines the task as video and a second model turns that video into motions; 1X reports success rates from 30 real trials per task and says dexterous tasks such as pouring and drawing remain hard.",
      "date": "2026-01",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/world-model-self-learning",
        "title": "1X World Model | From Video to Action: A New Way Robots Learn",
        "type": "blog",
        "date": "2026-01-12",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "1X Technologies",
      "what": "1X says NEO robots in its own factory collect real-world data and use the 1X World Model to learn practical tasks such as stocking parts for assembly technicians and basic warehousing (internal use).",
      "date": "2026-04",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/neo-factory",
        "title": "NEO Factory | Building Your NEO",
        "type": "blog",
        "date": "2026-04-30",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway offers a policy model service in which its GWM-1 world model predicts robot actions from live camera observations after custom fine-tuning for a customer's hardware, and also licenses GWM-1 as a backbone for custom policy models.",
      "date": "2025-12",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/product/robotics",
        "title": "Runway Robotics: Build smarter robots with general world models",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA",
      "what": "NVIDIA released Cosmos 3 in May 2026 as an open model that can generate robot actions as well as video, and in July 2026 added Cosmos 3 Edge, a 4-billion-parameter version that generates robot actions on NVIDIA Jetson edge computers.",
      "date": "2026-07",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-3-the-open-frontier-foundation-model-for-physical-ai",
        "title": "NVIDIA Launches Cosmos 3, the Open Frontier Foundation Model for Physical AI",
        "type": "press-release",
        "date": "2026-05-31",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/japans-robotics-and-manufacturing-leaders-build-on-nvidia-cosmos-to-advance-physical-ai-frontier",
        "title": "Japan's Robotics and Manufacturing Leaders Build on NVIDIA Cosmos to Advance Physical AI Frontier",
        "type": "press-release",
        "date": "2026-07-15",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA (DreamZero)",
      "what": "NVIDIA's DreamZero, a 14B video-based 'world action model', reports over 2x better generalization to new tasks and environments than vision-language-action models in real-robot tests, running closed-loop control at 7Hz.",
      "date": "2026-02",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2602.15922",
        "title": "World Action Models are Zero-shot Policies",
        "type": "paper",
        "date": "2026-02-17",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA (Cosmos Policy)",
      "what": "Cosmos Policy fine-tunes the Cosmos-Predict2 video model into a robot policy that also predicts future images and values for planning, and reports 98.5% average success on LIBERO and 67.1% on RoboCasa plus the highest average score among compared methods on real two-arm tasks.",
      "date": "2026-01",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2601.16163",
        "title": "Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning",
        "type": "paper",
        "date": "2026-01-22",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "AgiBot (Genie Envisioner)",
      "what": "AgiBot's Genie Envisioner trains a video world model on robot data and adds an action decoder (GE-Act) that controls robots including its AgiBot G1, with a world simulator (GE-Sim) for closed-loop policy development.",
      "date": "2025-08",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2508.05635",
        "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
        "type": "paper",
        "date": "2025-08-07",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Meta FAIR (V-JEPA 2-AC)",
      "what": "Meta post-trained its V-JEPA 2 model on less than 62 hours of robot video from the DROID dataset and used it to plan pick-and-place on Franka arms in two labs with no data from those labs, reporting 65% to 80% success with new objects.",
      "date": "2025-06",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2506.09985",
        "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
        "type": "paper",
        "date": "2025-06-11",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://ai.meta.com/blog/v-jepa-2-world-model-benchmarks/",
        "title": "Introducing V-JEPA 2 (Meta AI blog)",
        "type": "blog",
        "date": "2025-06-11",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "ByteDance Research (GR-2)",
      "what": "ByteDance pre-trained GR-2 on 38 million internet video clips to learn how the world changes, then fine-tuned it on robot data, and reports a 97.7% average success rate across more than 100 tasks.",
      "date": "2024-10",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2410.06158",
        "title": "GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation",
        "type": "paper",
        "date": "2024-10-08",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "New York University (DINO-WM)",
      "what": "DINO-WM predicts future image features instead of pixels and plans action sequences toward a goal image, and its authors report zero-shot solutions on six environments without expert demonstrations or reward models.",
      "date": "2024-11",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2411.04983",
        "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
        "type": "paper",
        "date": "2024-11-07",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "UC Berkeley; Google DeepMind and others (UniSim)",
      "what": "UniSim learned an interactive simulator from video and robot data and used it to train high-level and low-level policies that were then run on real robots without further training.",
      "date": "2023-10",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2310.06114",
        "title": "Learning Interactive Real-World Simulators",
        "type": "paper",
        "date": "2023-10-09",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "UC Berkeley (DayDreamer)",
      "what": "DayDreamer applied the Dreamer world-model algorithm to 4 physical robots and reports that a quadruped learned to roll over, stand up and walk from scratch in 1 hour without a simulator.",
      "date": "2022-06",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2206.14176",
        "title": "DayDreamer: World Models for Physical Robot Learning",
        "type": "paper",
        "date": "2022-06-28",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Large video models are slow for real-time control: NVIDIA needed model and system changes to run its 14B DreamZero model at 7Hz.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2602.15922",
        "title": "World Action Models are Zero-shot Policies",
        "type": "paper",
        "date": "2026-02-17",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Models trained on ordinary single-camera video have weak 3D understanding, so 1X reports that its robot can undershoot or overshoot even when the imagined video shows success.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/world-model-self-learning",
        "title": "1X World Model | From Video to Action: A New Way Robots Learn",
        "type": "blog",
        "date": "2026-01-12",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "Fine-grained tasks remain hard: 1X reports that pouring and drawing are still challenging for its world-model policy.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/world-model-self-learning",
        "title": "1X World Model | From Video to Action: A New Way Robots Learn",
        "type": "blog",
        "date": "2026-01-12",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "1X takes NEO orders at $20,000 for early-access ownership or $499 per month, says US deliveries start in 2026, and says it booked 10,000 NEOs in 5 days after the October 2025 launch; it set up a dedicated 1X World Model Lab in June 2026.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/order",
        "title": "Order NEO",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://www.1x.tech/discover/neo-factory",
        "title": "NEO Factory | Building Your NEO",
        "type": "blog",
        "date": "2026-04-30",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://www.1x.tech/discover/1x-world-model-lab",
        "title": "1X Launches World Model Lab to Scale Humanoid Intelligence",
        "type": "blog",
        "date": "2026-06-04",
        "accessed": "2026-10-10"
       }
      ]
     }
    ]
   },
   {
    "id": "driving-simulation",
    "name": "Driving simulation and scenario generation",
    "plain": "The world model generates realistic driving video and other sensor data, including rare or dangerous situations, and can respond to what the self-driving software does. Developers use it to train and test driving software on situations that are too rare, costly or risky to wait for on real roads.",
    "maturity": "product",
    "maturity_basis": "NVIDIA, Foretellix, Helm.ai and Decart sell world-model-based driving simulation or data tools to outside developers, and Waymo, Wayve, XPeng, Li Auto, Huawei, Pony.ai and comma.ai describe world models as part of their own production training or testing pipelines.",
    "evidence": [
     {
      "org": "Foretellix; NVIDIA",
      "what": "Foretellix added NVIDIA's Cosmos Transfer world model to its Foretify toolchain so that generated sensor data for training and testing self-driving software can vary in weather, lighting and location.",
      "date": "2025-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.foretellix.com/data-automation-toolchain-for-ai-powered-av-development/",
        "title": "Foretellix Expands Data Automation Toolchain for AI-Powered AV Development with Breakthrough Simulation Capabilities Using NVIDIA Omniverse and Cosmos Transfer",
        "type": "blog",
        "date": "2025-03-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-announces-major-release-of-cosmos-world-foundation-models-and-physical-ai-data-tools",
        "title": "NVIDIA Announces Major Release of Cosmos World Foundation Models and Physical AI Data Tools",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Uber; NVIDIA",
      "what": "Uber says it and NVIDIA are building a robotaxi data factory powered by the NVIDIA Cosmos platform, and that Uber will collect more than 3 million hours of robotaxi driving data for training and validation.",
      "date": "2025-10",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://investor.uber.com/news-events/news/press-release-details/2025/Uber-to-Deploy-One-of-the-Worlds-Largest-Networks-of-Autonomous-Vehicles-Powered-by-NVIDIA-AI-Architecture/default.aspx",
        "title": "Uber to Deploy One of the World's Largest Networks of Autonomous Vehicles, Powered by NVIDIA AI Architecture",
        "type": "press-release",
        "date": "2025-10-28",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Nexar; Oxa; Parallel Domain; Li Auto; NVIDIA",
      "what": "NVIDIA says Nexar and Oxa use Cosmos Predict for their driving systems and Parallel Domain uses its Cosmos-based simulation blueprint (March 2025), and lists Li Auto as building on Cosmos for autonomous vehicles (May 2026).",
      "date": "2026-05",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-announces-major-release-of-cosmos-world-foundation-models-and-physical-ai-data-tools",
        "title": "NVIDIA Announces Major Release of Cosmos World Foundation Models and Physical AI Data Tools",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-3-the-open-frontier-foundation-model-for-physical-ai",
        "title": "NVIDIA Launches Cosmos 3, the Open Frontier Foundation Model for Physical AI",
        "type": "press-release",
        "date": "2026-05-31",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA",
      "what": "NVIDIA offers Cosmos world models, including Cosmos-Drive models specialised for driving, to generate rare driving scenarios for perception and driving-policy training; its paper reports gains on 3D lane detection, 3D object detection and driving policy learning.",
      "date": "2025-06",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2506.09042",
        "title": "Cosmos-Drive-Dreams: Scalable Synthetic Driving Data Generation with World Foundation Models",
        "type": "paper",
        "date": "2025-06-10",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-launches-cosmos-world-foundation-model-platform-to-accelerate-physical-ai-development",
        "title": "NVIDIA Launches Cosmos World Foundation Model Platform to Accelerate Physical AI Development",
        "type": "press-release",
        "date": "2025-01-06",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Decart",
      "what": "Decart launched Oasis 3 in June 2026 as an interactive world model for training and testing robots, starting with self-driving, available through its API; its price list shows 'Oasis 3 Preview' as a real-time driving simulator at $0.02 per second at 720p.",
      "date": "2026-06",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
        "title": "Introducing Oasis 3: First Interactive World Model for Physical AI",
        "type": "blog",
        "date": "2026-06-10",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://docs.platform.decart.ai/getting-started/pricing",
        "title": "Pricing - Decart API Platform Documentation",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Helm.ai",
      "what": "Helm.ai sells generative simulation models to automakers and in May 2026 launched GenSim-3, which restyles real driving video across six surround cameras, and VidGen-3, which generates driving sequences from scratch; we found no named customers on its pages.",
      "date": "2026-05",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://helm.ai/post/vidgen-3-and-gensim-3",
        "title": "Helm.ai Sets New Full HD (2MP) Standard for Generative Simulation",
        "type": "press-release",
        "date": "2026-05-27",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Waymo; Google DeepMind",
      "what": "Waymo built the Waymo World Model on Google DeepMind's Genie 3 to generate camera and lidar data for its driving simulator, including rare events such as a tornado or an elephant on the road, and its engineers can change scenes with text prompts, driving inputs and layouts (internal use).",
      "date": "2026-02",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation",
        "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
        "type": "blog",
        "date": "2026-02-06",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Wayve",
      "what": "Wayve's GAIA-4 runs its AI Driver inside the world model in closed loop, generates radar as well as camera data, and is used to ask what the AI Driver would have done where a safety driver took over; Wayve reports that a 'world-on-rails' training mode made the model preserve the recorded scene 2.5x more faithfully (internal use).",
      "date": "2026-08",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/thinking/gaia-4/",
        "title": "GAIA-4: Multimodal World Models Powering Closed-Loop Simulation for Safe and Scalable Autonomy",
        "type": "blog",
        "date": "2026-08-03",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Wayve",
      "what": "Wayve launched GAIA-3, a 15-billion-parameter world model for evaluating driving software, and says early studies show its simulated tests closely mirror real-world driving results and cut synthetic-test rejection rates fivefold.",
      "date": "2025-12",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/wayve-launches-gaia3/",
        "title": "Wayve launches GAIA-3, advancing world models from simulation to evaluation",
        "type": "press-release",
        "date": "2025-12-02",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "XPeng",
      "what": "XPeng says its X-World world model supports its closed-loop simulation testing, online reinforcement learning and data generation, and that its driving simulation scenarios grew from 30,000 to more than 500,000 in a year, with daily simulated mileage equal to 30 million kilometers of real driving (internal use).",
      "date": "2026-04",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.xpeng.com/news/019dd72da86c9dd703de8a0282290002",
        "title": "XPENG Releases World Model Technical Report, Powering VLA 2.0 Model R&D and Verification",
        "type": "press-release",
        "date": "2026-04-29",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://arxiv.org/abs/2603.19979",
        "title": "X-World: Controllable Ego-Centric Multi-Camera World Models for Scalable End-to-End Driving",
        "type": "paper",
        "date": "2026-03-20",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Li Auto",
      "what": "Li Auto says it uses a cloud-based generative world model to create simulated scenarios that train its in-car VLA Driver model through reinforcement learning, and that the VLA Driver was rolled out across its Li AD Max models from September 2025 (internal use).",
      "date": "2026-03",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1791706/000110465926026803/tm268559d1_ex99-2.pdf",
        "title": "Li Auto Inc. Annual Results Announcement for the Year Ended December 31, 2025 (Form 6-K exhibit 99.2)",
        "type": "press-release",
        "date": "2026-03",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Huawei (Qiankun)",
      "what": "Huawei says the cloud 'World Engine' in its ADS 5 driving system (WEWA 2.0 architecture) generates multi-agent scenarios for online reinforcement learning and raises training intensity and training efficiency by 10 times each (internal use).",
      "date": "2026-04",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://auto.huawei.com/cn/news/2026/2026-04-23-jishu",
        "title": "2026 华为乾崑技术大会在京举行 (2026 Huawei Qiankun Technology Conference)",
        "type": "press-release",
        "date": "2026-04-23",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://auto.huawei.com/cn/ads",
        "title": "乾崑智驾 ADS 5 product page",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Pony.ai",
      "what": "Pony.ai describes PonyWorld as its proprietary world model and reinforcement-learning training system, built since 2020, and says PonyWorld 2.0 is applied across its driverless robotaxi fleet and R&D to find weak spots and target training on the hardest cases (internal use).",
      "date": "2026-04",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.prnewswire.com/news-releases/ponyai-launches-ponyworld-2-0--a-self-improving-physical-ai-engine-for-autonomous-driving-302739012.html",
        "title": "Pony.ai Launches PonyWorld 2.0, a Self-Improving Physical AI Engine for Autonomous Driving",
        "type": "press-release",
        "date": "2026-04-10",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "comma.ai",
      "what": "comma.ai says openpilot 0.11 ships a driving model trained with both videos and plans generated by a world model, which it calls the first real-world robotics agent shipped to users that was fully trained in a learned simulation (internal use of the world model; the driving model ships to users).",
      "date": "2026-03",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://blog.comma.ai/011release/",
        "title": "openpilot 0.11",
        "type": "blog",
        "date": "2026-03-17",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://arxiv.org/abs/2504.19077",
        "title": "Learning to Drive from a World Model",
        "type": "paper",
        "date": "2025-04-27",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Tesla",
      "what": "Tesla's head of AI software described a 'neural world simulator' that generates multi-camera video in response to the driving software, used to train and test its FSD software and its Optimus robot; we found this only in secondary coverage of his ICCV 2025 talk.",
      "date": "2025-10",
      "kind": "pilot",
      "level": "reported",
      "sources": [
       {
        "url": "https://www.humanoidsdaily.com/news/tesla-ai-chief-details-unified-world-simulator-for-fsd-and-optimus",
        "title": "Tesla AI Chief Details Unified 'World Simulator' for FSD and Optimus (Humanoids Daily)",
        "type": "secondary",
        "date": "2025-10-24",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Wayve (GAIA-2)",
      "what": "Wayve's GAIA-2 paper describes a world model that generates multi-camera driving video controlled by vehicle motion, other road users, weather and road layout, across the UK, US and Germany.",
      "date": "2025-03",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2503.20523",
        "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
        "type": "paper",
        "date": "2025-03-26",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "A world model can invent a more forgiving future, for example leaving out a car that really entered a junction, which changes how dangerous a scenario looks; Wayve had to add a constraint to keep other road users on their recorded paths for safety testing.",
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/thinking/gaia-4/",
        "title": "GAIA-4: Multimodal World Models Powering Closed-Loop Simulation for Safe and Scalable Autonomy",
        "type": "blog",
        "date": "2026-08-03",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Methods to show that world-model test results can count as safety evidence are still being built: Wayve is working with the University of Warwick on DriveSafeSim, a UK government-funded project to validate generative world models for safety evaluation.",
      "level": "verified",
      "sources": [
       {
        "url": "https://wayve.ai/press/wayve-launches-gaia3/",
        "title": "Wayve launches GAIA-3, advancing world models from simulation to evaluation",
        "type": "press-release",
        "date": "2025-12-02",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Running video world models at fleet scale is costly; XPeng built a separate accelerator (X-Cache) that it says makes the model's denoising step up to 2.7 times faster.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.xpeng.com/pressroom/news/019e0199813b9dd703de8a02822900a1",
        "title": "XPENG Unveils the World Model Accelerator X-Cache, Which Requires No Training, Is Plug-and-Play, and Boosts Inference Speed by 2.7 Times",
        "type": "press-release",
        "date": "2026-05-06",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Most results are company claims without shared benchmarks: Waymo's announcement gives no measured accuracy figures, and Wayve's 'closely mirrors' statement is not backed by a published statistic.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation",
        "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
        "type": "blog",
        "date": "2026-02-06",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://wayve.ai/press/wayve-launches-gaia3/",
        "title": "Wayve launches GAIA-3, advancing world models from simulation to evaluation",
        "type": "press-release",
        "date": "2025-12-02",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "Uber says it will collect more than 3 million hours of robotaxi driving data for a data factory built with NVIDIA's Cosmos platform.",
      "level": "verified",
      "sources": [
       {
        "url": "https://investor.uber.com/news-events/news/press-release-details/2025/Uber-to-Deploy-One-of-the-Worlds-Largest-Networks-of-Autonomous-Vehicles-Powered-by-NVIDIA-AI-Architecture/default.aspx",
        "title": "Uber to Deploy One of the World's Largest Networks of Autonomous Vehicles, Powered by NVIDIA AI Architecture",
        "type": "press-release",
        "date": "2025-10-28",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "what": "Several carmakers now name world models in investor and product materials: Li Auto in its 2025 annual results, NIO in its January 2026 delivery update, Huawei for ADS 5 and XPeng for VLA 2.0.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1791706/000110465926026803/tm268559d1_ex99-2.pdf",
        "title": "Li Auto Inc. Annual Results Announcement for the Year Ended December 31, 2025 (Form 6-K exhibit 99.2)",
        "type": "press-release",
        "date": "2026-03",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1736541/000110465926008827/tm264763d1_ex99-1.htm",
        "title": "NIO Inc. Provides January 2026 Delivery Update (Form 6-K exhibit 99.1)",
        "type": "press-release",
        "date": "2026-02-01",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://auto.huawei.com/cn/news/2026/2026-04-23-jishu",
        "title": "2026 华为乾崑技术大会在京举行 (2026 Huawei Qiankun Technology Conference)",
        "type": "press-release",
        "date": "2026-04-23",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.xpeng.com/news/019dd72da86c9dd703de8a0282290002",
        "title": "XPENG Releases World Model Technical Report, Powering VLA 2.0 Model R&D and Verification",
        "type": "press-release",
        "date": "2026-04-29",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "driving-onboard",
    "name": "In-car driving models",
    "plain": "A model running in the car predicts how the road scene will develop and uses that prediction to plan the car's movements. Carmakers want this so driver-assistance systems react earlier and more smoothly in complex traffic.",
    "maturity": "product",
    "maturity_basis": "NIO ships driver-assistance software it calls the NIO WorldModel to more than 460,000 cars, and Huawei sells ADS 5 with a vehicle-side 'World Action Model'; neither company publishes enough technical detail to show how these differ from other end-to-end driving models.",
    "evidence": [
     {
      "org": "NIO",
      "what": "NIO says that on January 28, 2026 it rolled out the latest version of its NIO WorldModel (NWM) driver-assistance software to over 460,000 cars with its Banyan system, adding closed-loop reinforcement learning for urban and highway driving.",
      "date": "2026-01",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1736541/000110465926008827/tm264763d1_ex99-1.htm",
        "title": "NIO Inc. Provides January 2026 Delivery Update (Form 6-K exhibit 99.1)",
        "type": "press-release",
        "date": "2026-02-01",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Huawei (Qiankun)",
      "what": "Huawei's ADS 5 driver-assistance system pairs a cloud 'World Engine' with a vehicle-side 'World Action Model', and Huawei says the in-car model's new risk-field method can cut collision risk by 50%; Huawei states that ADS in passenger cars is assisted driving that cannot replace the driver.",
      "date": "2026-04",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://auto.huawei.com/cn/news/2026/2026-04-23-jishu",
        "title": "2026 华为乾崑技术大会在京举行 (2026 Huawei Qiankun Technology Conference)",
        "type": "press-release",
        "date": "2026-04-23",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://auto.huawei.com/cn/ads",
        "title": "乾崑智驾 ADS 5 product page",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "XPeng",
      "what": "XPeng presented X-Mind, a research framework that embeds a predictive world model inside the in-car driving model so the car imagines a compact sketch of the near future before planning its path; XPeng does not say it is in production cars yet.",
      "date": "2026-06",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.xpeng.com/news/019f12539bff9f1220b48a028223000e",
        "title": "X-Mind: Empowering Autonomous Driving with a Future-Foresight Brain",
        "type": "press-release",
        "date": "2026-06-29",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Predicting full future video inside the car is too heavy for real-time use, so XPeng's X-Mind predicts a compact abstract sketch of the future instead of raw images.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.xpeng.com/news/019f12539bff9f1220b48a028223000e",
        "title": "X-Mind: Empowering Autonomous Driving with a Future-Foresight Brain",
        "type": "press-release",
        "date": "2026-06-29",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Carmakers use 'world model' as a product name and publish few technical details, so outsiders cannot check what the in-car model predicts or how it is tested.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1736541/000110465926008827/tm264763d1_ex99-1.htm",
        "title": "NIO Inc. Provides January 2026 Delivery Update (Form 6-K exhibit 99.1)",
        "type": "press-release",
        "date": "2026-02-01",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://auto.huawei.com/cn/ads",
        "title": "乾崑智驾 ADS 5 product page",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "These systems are sold as driver assistance, so the driver stays responsible; Huawei's own page says ADS cannot replace the driver or handle all road, weather and traffic conditions.",
      "level": "verified",
      "sources": [
       {
        "url": "https://auto.huawei.com/cn/ads",
        "title": "乾崑智驾 ADS 5 product page",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "NIO reports that its world-model-branded software reached over 460,000 Banyan-system cars in January 2026.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.sec.gov/Archives/edgar/data/1736541/000110465926008827/tm264763d1_ex99-1.htm",
        "title": "NIO Inc. Provides January 2026 Delivery Update (Form 6-K exhibit 99.1)",
        "type": "press-release",
        "date": "2026-02-01",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "games-interactive",
    "name": "Games and interactive media",
    "plain": "The world model draws a playable world frame by frame in response to a player's keyboard, controller or text input, so a world can be explored without a game engine or hand-built assets. Players get worlds made on request, and game teams can try out ideas before building them the usual way.",
    "maturity": "product",
    "maturity_basis": "Google sells access to Project Genie inside its Google AI Ultra subscription in more than 140 countries, and Odyssey and Decart offer paid world-model APIs, but every source we found describes playable worlds as experiments, and no shipped commercial game built with a world model was found.",
    "evidence": [
     {
      "org": "Google DeepMind; Google",
      "what": "Google launched Project Genie, a research prototype built on the Genie 3 world model that lets users create and explore interactive worlds, for Google AI Ultra subscribers in the U.S. in January 2026; Google's plan page now says it is available in more than 140 countries, only with a Google AI Ultra plan.",
      "date": "2026-01",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/project-genie/",
        "title": "Project Genie: Experimenting with infinite, interactive worlds",
        "type": "blog",
        "date": "2026-01-29",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://gemini.google/subscriptions/",
        "title": "Google AI plans and subscriptions",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Odyssey",
      "what": "Odyssey streams interactive AI-generated video that responds to typed input (Odyssey-2, October 2025), offers its models to developers through an API for uses it lists as robotics, gaming, education and defense, and in May 2026 showed Agora-1, a world model in which up to four players share one generated world.",
      "date": "2026-05",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/introducing-odyssey-2",
        "title": "Introducing Odyssey-2: A General-Purpose World Model",
        "type": "blog",
        "date": "2025-10-27",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://odyssey.systems/investment-from-nvidia-and-samsung",
        "title": "Odyssey Announces Investment from NVentures and Samsung Next",
        "type": "blog",
        "date": "2026-02-12",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://odyssey.systems/introducing-agora-1",
        "title": "Agora-1: The Multi-Agent World Model",
        "type": "blog",
        "date": "2026-05-18",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Decart; Etched",
      "what": "Decart and Etched released Oasis in October 2024, a playable open-world game generated frame by frame from keyboard and mouse input with no game engine, with open code and weights; Decart says Oasis began as a gaming world model and its third version (June 2026) now targets robots and self-driving.",
      "date": "2024-10",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://oasis-model.github.io/",
        "title": "Oasis: A Universe in a Transformer",
        "type": "site",
        "date": "2024-10-31",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.decart.ai/publications/oasis-interactive-ai-video-game-model",
        "title": "Oasis: A Universe in a Transformer (Decart)",
        "type": "blog",
        "date": "2024-10-31",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.decart.ai/publications/introducing-oasis-3-first-interactive-world-model-for-physical-ai",
        "title": "Introducing Oasis 3: First Interactive World Model for Physical AI",
        "type": "blog",
        "date": "2026-06-10",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Microsoft Research; Xbox",
      "what": "Microsoft put an AI-generated, playable version of Quake II gameplay into Copilot Labs, run by its WHAMM world model, and states that the model only remembers 0.9 seconds of gameplay (9 frames at 10fps), so objects out of view for longer are forgotten.",
      "date": "2025-04",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.microsoft.com/en-us/research/articles/whamm-real-time-world-modelling-of-interactive-environments/",
        "title": "WHAMM! Real-time world modelling of interactive environments",
        "type": "blog",
        "date": "2025-04-04",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "General Intuition; Kyutai; Epic Games",
      "what": "General Intuition and Kyutai, in collaboration with Epic Games, built MIRA, a 5B-parameter model that simulates Rocket League for four players at once at 20 fps, and state that the technology is a demo and is not being used to develop Rocket League.",
      "date": "2026-07",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://mira-wm.com/blog-post/",
        "title": "MIRA: Multiplayer Interactive World Models with Representation Autoencoders",
        "type": "blog",
        "date": "2026-07",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://arxiv.org/abs/2607.05352",
        "title": "Multiplayer Interactive World Models with Representation Autoencoders",
        "type": "paper",
        "date": "2026-07",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway's GWM Worlds model turns a still scene into an explorable space generated in real time, and Runway lists gaming, education, training agents and VR among its uses; access is by request form.",
      "date": "2025-12",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Microsoft Research; Ninja Theory (Xbox Game Studios)",
      "what": "Microsoft Research and Ninja Theory trained Muse, a 'World and Human Action Model', on more than 1 billion images and controller actions (over 7 years of gameplay) from the game Bleeding Edge to support gameplay ideation, published it in Nature, and released the weights; Xbox said it is exploring Muse for bringing older games to new devices.",
      "date": "2025-02",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.microsoft.com/en-us/research/blog/introducing-muse-our-first-generative-ai-model-designed-for-gameplay-ideation/",
        "title": "Introducing Muse: Our first generative AI model designed for gameplay ideation",
        "type": "blog",
        "date": "2025-02-19",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://news.xbox.com/en-us/2025/02/19/muse-ai-xbox-empowering-creators-and-players/",
        "title": "Empowering Creators and Players With Muse, a Generative AI Model for Gameplay",
        "type": "blog",
        "date": "2025-02-19",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Tencent Hunyuan",
      "what": "Tencent's Hunyuan-GameCraft generates game video that follows keyboard and mouse input and was trained on over one million gameplay recordings from over 100 AAA games.",
      "date": "2025-06",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2506.17201",
        "title": "Hunyuan-GameCraft: High-dynamic Interactive Game Video Generation with Hybrid History Condition",
        "type": "paper",
        "date": "2025-06-20",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://github.com/Tencent-Hunyuan/Hunyuan-GameCraft-1.0",
        "title": "Tencent-Hunyuan/Hunyuan-GameCraft-1.0 (GitHub)",
        "type": "repo",
        "date": "2025-08",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Skywork AI",
      "what": "Skywork's open-source Matrix-Game 2.0 generates interactive, minute-long video at 25 FPS from mouse and keyboard input, trained on about 1200 hours of video produced in Unreal Engine and GTA5 environments; its GitHub repository now hosts Matrix-Game 3.0.",
      "date": "2025-08",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2508.13009",
        "title": "Matrix-Game 2.0: An Open-Source, Real-Time, and Streaming Interactive World Model",
        "type": "paper",
        "date": "2025-08-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://github.com/SkyworkAI/Matrix-Game",
        "title": "SkyworkAI/Matrix-Game (GitHub)",
        "type": "repo",
        "date": "2025-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Google Research; Tel Aviv University; Google DeepMind (GameNGen)",
      "what": "GameNGen simulated the game DOOM with a neural model at 20 frames per second on a single TPU, and human raters were only slightly better than chance at telling short real clips from simulated ones.",
      "date": "2024-08",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2408.14837",
        "title": "Diffusion Models Are Real-Time Game Engines",
        "type": "paper",
        "date": "2024-08-27",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://gamengen.github.io/",
        "title": "GameNGen project page",
        "type": "site",
        "date": "2024-08",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Sessions are short and memory is limited: Project Genie generations are limited to 60 seconds, Genie 3 supports a few minutes of continuous interaction, and Microsoft's WHAMM remembers 0.9 seconds of play.",
      "level": "verified",
      "sources": [
       {
        "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/project-genie/",
        "title": "Project Genie: Experimenting with infinite, interactive worlds",
        "type": "blog",
        "date": "2026-01-29",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.microsoft.com/en-us/research/articles/whamm-real-time-world-modelling-of-interactive-environments/",
        "title": "WHAMM! Real-time world modelling of interactive environments",
        "type": "blog",
        "date": "2025-04-04",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Control and rules are loose: Google says Project Genie worlds may not follow prompts or real-world physics and characters can be less controllable or lag, and Genie 3 has a limited set of actions, weak multi-agent interaction and unreliable text.",
      "level": "verified",
      "sources": [
       {
        "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/project-genie/",
        "title": "Project Genie: Experimenting with infinite, interactive worlds",
        "type": "blog",
        "date": "2026-01-29",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Several game world models were trained on footage of commercial games owned by others (Hunyuan-GameCraft on recordings from over 100 AAA games, Matrix-Game 2.0 on GTA5 environments), and the papers do not say whether the game owners agreed; Microsoft trained Muse on its own game.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2506.17201",
        "title": "Hunyuan-GameCraft: High-dynamic Interactive Game Video Generation with Hybrid History Condition",
        "type": "paper",
        "date": "2025-06-20",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://arxiv.org/abs/2508.13009",
        "title": "Matrix-Game 2.0: An Open-Source, Real-Time, and Streaming Interactive World Model",
        "type": "paper",
        "date": "2025-08-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.microsoft.com/en-us/research/blog/introducing-muse-our-first-generative-ai-model-designed-for-gameplay-ideation/",
        "title": "Introducing Muse: Our first generative AI model designed for gameplay ideation",
        "type": "blog",
        "date": "2025-02-19",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "Odyssey announced a $310 million Series B at a $1.45 billion valuation in June 2026, with Amazon, GV, AMD Ventures, EQT and IQT among the investors.",
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/our-series-b",
        "title": "Our $310 Million Fundraise to Accelerate World Simulation",
        "type": "blog",
        "date": "2026-06-17",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "what": "Decart announced a $300M round led by Radical Ventures in May 2026 and names Amazon as one of its strategic customers.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.decart.ai/publications/decart-raises-300m-tech-leaders-back-the-company-as-both-customers-and-investors",
        "title": "Decart Raises $300M: Tech Leaders Back the Company as Both Customers and Investors",
        "type": "press-release",
        "date": "2026-05-18",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "what": "Google AI Ultra, the only plan that includes Project Genie, is listed at $99.99 per month and $199.99 per month for higher usage limits.",
      "level": "verified",
      "sources": [
       {
        "url": "https://gemini.google/subscriptions/",
        "title": "Google AI plans and subscriptions",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "what": "TechCrunch reported in June 2026 that General Intuition, which trains agents on video game clips, raised $320 million at a $2.3 billion valuation, bringing its disclosed funding to $454 million.",
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2026/06/25/general-intuitions-2-3b-bet-that-video-games-can-train-ai-agents-for-the-real-world/",
        "title": "General Intuition's $2.3B bet that video games can train AI agents for the real world (TechCrunch)",
        "type": "secondary",
        "date": "2026-06-25",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "film-vfx",
    "name": "Film and visual effects",
    "plain": "Video and world models generate shots, backgrounds, previews of scenes or 3D sets from text, images or rough footage. Studios want this to try ideas and make visual material with less filming, set building and manual effects work.",
    "maturity": "product",
    "maturity_basis": "Runway sells its models to studios and names Lionsgate, AMC Networks and the makers of Amazon's House of David as users, and Disney licensed its characters to OpenAI's Sora; most of these tools are video generators that their makers describe as early world models.",
    "evidence": [
     {
      "org": "Lionsgate; Runway",
      "what": "Lionsgate and Runway agreed in September 2024 to train a custom video model on Lionsgate's catalog for its filmmakers' pre- and post-production, and in June 2026 expanded the deal: Lionsgate took an equity interest in Runway and the two will co-develop new series using Runway's models.",
      "date": "2026-06",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/news/runway-partners-with-lionsgate",
        "title": "Runway Partners with Lionsgate",
        "type": "press-release",
        "date": "2024-09-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://runway.com/news/company-news/runway-and-lionsgate-expand-partnership",
        "title": "Runway and Lionsgate Expand Partnership",
        "type": "press-release",
        "date": "2026-06-11",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "AMC Networks; Runway",
      "what": "AMC Networks partnered with Runway to use Runway's models in marketing and TV development, including visual concepts for new series and special effects ideation.",
      "date": "2025-06",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/news/runway-amc-partnership",
        "title": "Runway Partners with AMC Networks Across Marketing and TV Development",
        "type": "press-release",
        "date": "2025-06-04",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "The Wonder Project; Vū Technologies; Amazon MGM Studios; Runway",
      "what": "Runway's customer story says the team behind the Amazon series House of David used Runway to create complex scenes in Episode Six and lists 5 months of post-production time saved; these figures come from Runway's own page.",
      "date": "",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/customers/how-house-of-david-used-runway-to-become-amazons-latest-hit-series",
        "title": "How House of David Used Runway to Become Amazon's Latest Hit Series",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "The Walt Disney Company; OpenAI",
      "what": "Disney signed a three-year licence letting OpenAI's Sora video model generate short fan videos with more than 200 Disney, Marvel, Pixar and Star Wars characters, some to be streamed on Disney+, and agreed a $1 billion equity investment in OpenAI; the deal excludes talent likenesses and voices.",
      "date": "2025-12",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://thewaltdisneycompany.com/press-releases/the-walt-disney-company-and-openai-reach-landmark-agreement-to-bring-beloved-characters-from-across-disneys-brands-to-sora/",
        "title": "The Walt Disney Company and OpenAI Reach Landmark Agreement to Bring Beloved Characters From Across Disney's Brands to Sora",
        "type": "press-release",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "HTC VIVE Mars; World Labs",
      "what": "World Labs and HTC's VIVE Mars camera-tracking team used Marble to turn a single image or text prompt into a 3D environment for virtual production film shoots, according to a World Labs case study.",
      "date": "2025-11",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/vive-mars",
        "title": "From Image to Immersive: VIVE Mars x Marble",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway started a research programme on 'general world models' in December 2023, describing video generators such as its Gen-2 as early and limited world models, and in December 2025 released GWM-1, built on its Gen-4.5 video model.",
      "date": "2025-12",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-general-world-models",
        "title": "Introducing General World Models",
        "type": "blog",
        "date": "2023-12-11",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Odyssey",
      "what": "Odyssey launched Explorer in December 2024 as a world model for film and gaming and added Pixar co-founder Ed Catmull to its board.",
      "date": "2024-12",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/introducing-explorer",
        "title": "World Models for Film, Gaming, and Beyond",
        "type": "blog",
        "date": "2024-12-18",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Rights limit what can be generated: Disney's licence to OpenAI excludes talent likenesses and voices, and studios work with custom or licensed models such as Lionsgate's model trained on its own catalog.",
      "level": "verified",
      "sources": [
       {
        "url": "https://thewaltdisneycompany.com/press-releases/the-walt-disney-company-and-openai-reach-landmark-agreement-to-bring-beloved-characters-from-across-disneys-brands-to-sora/",
        "title": "The Walt Disney Company and OpenAI Reach Landmark Agreement to Bring Beloved Characters From Across Disney's Brands to Sora",
        "type": "press-release",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://runway.com/news/runway-partners-with-lionsgate",
        "title": "Runway Partners with Lionsgate",
        "type": "press-release",
        "date": "2024-09-18",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Generated video drifts between takes: World Labs' film case study says generative video tools shift environments and lighting between shots and lose spatial logic, which is why creators add fixed 3D worlds.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/creative-film",
        "title": "Framing Worlds: How Marble Helps Creators Bring Consistency to AI Filmmaking",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "The 'world model' label in film is mostly the vendors' framing of video generators; Runway itself calls Gen-2 a very early and limited form of world model, so film uses say little about physical accuracy.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-general-world-models",
        "title": "Introducing General World Models",
        "type": "blog",
        "date": "2023-12-11",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "Disney agreed a $1 billion equity investment in OpenAI alongside its Sora licence (December 2025), and Lionsgate took an equity interest in Runway (June 2026).",
      "level": "verified",
      "sources": [
       {
        "url": "https://thewaltdisneycompany.com/press-releases/the-walt-disney-company-and-openai-reach-landmark-agreement-to-bring-beloved-characters-from-across-disneys-brands-to-sora/",
        "title": "The Walt Disney Company and OpenAI Reach Landmark Agreement to Bring Beloved Characters From Across Disney's Brands to Sora",
        "type": "press-release",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://runway.com/news/company-news/runway-and-lionsgate-expand-partnership",
        "title": "Runway and Lionsgate Expand Partnership",
        "type": "press-release",
        "date": "2026-06-11",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "3d-content-digital-twins",
    "name": "3D content and digital twins",
    "plain": "The world model builds a lasting 3D scene, such as a room, street or landscape, from text, photos or video, which can then be explored and exported to other 3D tools. Teams want this to get usable 3D environments, including copies of real places, without modelling every object by hand.",
    "maturity": "product",
    "maturity_basis": "World Labs sells Marble and a World API that generate exportable 3D worlds and publishes named case studies, and Tencent has released an open-source 3D world generator; uses for faithful copies of real sites (digital twins) are less developed.",
    "evidence": [
     {
      "org": "World Labs",
      "what": "World Labs made its Marble world model generally available in November 2025; it creates 3D worlds from text, images, video or rough 3D layouts and exports them as Gaussian splats, meshes or videos.",
      "date": "2025-11",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/blog/marble-world-model",
        "title": "Marble: A Multimodal World Model",
        "type": "blog",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "World Labs",
      "what": "World Labs launched the World API in January 2026 so that developers can generate explorable 3D worlds from text, images, panoramas, multi-view inputs and video inside their own products.",
      "date": "2026-01",
      "kind": "product",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/blog/announcing-the-world-api",
        "title": "Announcing the World API",
        "type": "blog",
        "date": "2026-01-21",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Lightwheel; World Labs",
      "what": "World Labs and Lightwheel describe a pipeline that uses Marble to generate the environment layer for real-to-simulation robot evaluation across many scenes.",
      "date": "2025-11",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/2-lightwheel",
        "title": "Generating the World Layer: How Lightwheel and World Labs Scale Robotics Evaluation",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "World Labs; architects and designers",
      "what": "World Labs publishes a case study on architects and interior designers using Marble to turn renderings and prompts into explorable 3D spaces.",
      "date": "2025-11",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/2-architecture-design",
        "title": "Reframing Space: How Marble Is Transforming Architectural and Interior Visualization",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Tencent Hunyuan",
      "what": "Tencent released HunyuanWorld 1.0 as open source in July 2025; it generates explorable 3D worlds from text or images with mesh export for standard graphics pipelines, and its GitHub repository had 2,944 stars on 2026-10-11.",
      "date": "2025-07",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2507.21809",
        "title": "HunyuanWorld 1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
        "type": "paper",
        "date": "2025-07-29",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://github.com/Tencent-Hunyuan/HunyuanWorld-1.0",
        "title": "Tencent-Hunyuan/HunyuanWorld-1.0 (GitHub)",
        "type": "repo",
        "date": "2025-07",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Meta (Reality Labs)",
      "what": "Meta Reality Labs' WorldGen research system turns text prompts into large, traversable 3D worlds that can be explored or edited in standard game engines.",
      "date": "2025-11",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2511.16825",
        "title": "WorldGen: From Text to Traversable and Interactive 3D Worlds",
        "type": "paper",
        "date": "2025-11-20",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Copies of real places need geographic accuracy that generative models do not yet guarantee; Google DeepMind says Genie 3 cannot simulate real-world locations with perfect geographic accuracy.",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "Methods trade off variety against 3D consistency: Tencent notes that video-based world generators lack 3D consistency while 3D-based methods are limited by scarce training data.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2507.21809",
        "title": "HunyuanWorld 1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
        "type": "paper",
        "date": "2025-07-29",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "World Labs raised $1 billion in February 2026 from investors including AMD, Autodesk, Fidelity, NVIDIA and Sea, and in September 2026 signed a definitive agreement to join AMD, expected to close by the end of 2026.",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/blog/funding-2026",
        "title": "World Labs Announces New Funding",
        "type": "blog",
        "date": "2026-02-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.worldlabs.ai/blog/amd-announcement",
        "title": "World Labs is Joining AMD",
        "type": "blog",
        "date": "2026-09-28",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "industrial-warehouse",
    "name": "Industrial and warehouse simulation",
    "plain": "World models generate or vary video of factories and warehouses, or let robots practise there in imagination, so companies can test robot fleets and new tasks before changing the real site. Most named factory projects still rely mainly on classic physics-based digital twins, with world models added for realistic visuals or robot learning.",
    "maturity": "pilot",
    "maturity_basis": "1X uses its world model to teach robots tasks in its own factory and several manufacturers are trying NVIDIA's Cosmos-linked tools, but the sources describe exploration and early adoption and give no results.",
    "evidence": [
     {
      "org": "1X Technologies",
      "what": "1X says NEO robots in its own factory use the 1X World Model to learn practical tasks such as stocking parts for assembly technicians and basic warehousing and logistics (internal use).",
      "date": "2026-04",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.1x.tech/discover/neo-factory",
        "title": "NEO Factory | Building Your NEO",
        "type": "blog",
        "date": "2026-04-30",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Schaeffler; Accenture; Hyundai Motor Group; Mercedes-Benz; KION Group; Dematic; NVIDIA",
      "what": "NVIDIA says Schaeffler and Accenture are starting to use its Mega blueprint to simulate fleets of Agility Robotics' Digit, Hyundai to simulate Boston Dynamics Atlas robots, Mercedes-Benz to simulate Apptronik Apollo robots, and KION, Dematic and Accenture to plan warehouse automation; NVIDIA calls these blueprints connected to Cosmos but does not say what part the world model plays for each user.",
      "date": "2025-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-omniverse-physical-ai-operating-system-expands-to-more-industries-and-partners",
        "title": "NVIDIA Omniverse Physical AI Operating System Expands to More Industries and Partners",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Berkshire Grey; Runway",
      "what": "Runway says it is building its GWM-Robotics world model with partners including the warehouse automation company Berkshire Grey and NVIDIA.",
      "date": "2026-02",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
        "title": "Accelerating Robot Policy Evaluation with General World Models",
        "type": "blog",
        "date": "2026-02-27",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Mujin; Kubota; Telexistence; NVIDIA",
      "what": "NVIDIA says Mujin is exploring Cosmos for industrial automation, Kubota is exploring Cosmos-based physical AI for farming, and Telexistence is exploring Cosmos for retail automation.",
      "date": "2026-07",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/japans-robotics-and-manufacturing-leaders-build-on-nvidia-cosmos-to-advance-physical-ai-frontier",
        "title": "Japan's Robotics and Manufacturing Leaders Build on NVIDIA Cosmos to Advance Physical AI Frontier",
        "type": "press-release",
        "date": "2026-07-15",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "The named factory and warehouse projects are described as exploring or starting to adopt, and none publishes measured results, so the benefit of world models over classic digital twins is not yet shown.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-omniverse-physical-ai-operating-system-expands-to-more-industries-and-partners",
        "title": "NVIDIA Omniverse Physical AI Operating System Expands to More Industries and Partners",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/japans-robotics-and-manufacturing-leaders-build-on-nvidia-cosmos-to-advance-physical-ai-frontier",
        "title": "Japan's Robotics and Manufacturing Leaders Build on NVIDIA Cosmos to Advance Physical AI Frontier",
        "type": "press-release",
        "date": "2026-07-15",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "text": "NVIDIA's Cosmos brand covers both world models and a vision-language model (Cosmos Reason), so a company 'using Cosmos' in a factory may not be using a world model at all.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-opens-portals-to-world-of-robotics-with-new-omniverse-libraries-cosmos-physical-ai-models-and-ai-computing-infrastructure",
        "title": "NVIDIA Opens Portals to World of Robotics With New Omniverse Libraries, Cosmos Physical AI Models and AI Computing Infrastructure",
        "type": "press-release",
        "date": "2025-08-11",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "market_signals": []
   },
   {
    "id": "ar-vr",
    "name": "AR and VR",
    "plain": "World models generate 3D or video worlds that people can view or walk through with a headset. This could make immersive scenes for entertainment, design or therapy without building every asset by hand.",
    "maturity": "pilot",
    "maturity_basis": "Named groups such as HTC's VIVERSE platform and researchers at the Champalimaud Foundation have used World Labs' Marble in immersive projects, but no headset maker ships world-model scene generation as a feature that we could confirm.",
    "evidence": [
     {
      "org": "HTC VIVERSE; World Labs",
      "what": "HTC's VIVERSE platform worked with World Labs to test turning Marble-generated 3D worlds into interactive experiences built by creators.",
      "date": "2025-11",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/viverse",
        "title": "Building Worlds Together: How VIVERSE and Marble Empowered Creators to Build Interactive 3D Worlds",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Champalimaud Foundation; King's College London; World Labs",
      "what": "Researchers at the Champalimaud Foundation and King's College London use Marble to turn short text descriptions into immersive, patient-specific scenes for exposure therapy for obsessive-compulsive disorder.",
      "date": "2025-11",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/case-studies/3-health-systems",
        "title": "Immersive Exposure: Generative Worlds for OCD Therapy",
        "type": "site",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "World Labs (showcase by VR developer Daniel Skaale)",
      "what": "A VR developer built 'Splat World', a real-time VR experience in Unity, from environments generated with Marble, according to World Labs' showcase.",
      "date": "",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://www.worldlabs.ai/labs/showcase/gaussian-splats-vr",
        "title": "Splat World (World Labs showcase)",
        "type": "site",
        "date": "undated",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Tencent Hunyuan",
      "what": "Tencent lists virtual reality among the applications of HunyuanWorld 1.0, which uses 360-degree panoramas as the basis of its generated worlds.",
      "date": "2025-07",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2507.21809",
        "title": "HunyuanWorld 1.0: Generating Immersive, Explorable, and Interactive 3D Worlds from Words or Pixels",
        "type": "paper",
        "date": "2025-07-29",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway lists VR and immersive experiences among the intended uses of GWM Worlds; this is a stated intention.",
      "date": "2025-12",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Frame-by-frame world models keep scenes consistent only for short periods (a few minutes for Genie 3), which is short for headset sessions; exported 3D worlds such as Marble's avoid this but are static scenes.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.worldlabs.ai/blog/marble-world-model",
        "title": "Marble: A Multimodal World Model",
        "type": "blog",
        "date": "2025-11-12",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": []
   },
   {
    "id": "education-training",
    "name": "Education and training simulators",
    "plain": "The world model would generate scenes that students or workers can explore or practise in, such as historical places or emergency situations, without building each training simulation by hand. The appeal is cheaper and more varied practice environments.",
    "maturity": "research",
    "maturity_basis": "Google DeepMind, Runway and Odyssey list education and training as intended uses, but in the pages we checked no school, training provider or employer is named as using a world model to train people; our search here was cut short by a tool limit (see notes).",
    "evidence": [
     {
      "org": "Google DeepMind",
      "what": "Google DeepMind wrote that Genie 3 could create new opportunities for education and training, helping students learn and experts gain experience; this is a stated intention, not a reported use.",
      "date": "2025-08",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Runway",
      "what": "Runway lists education among the uses of its GWM Worlds model and describes Runway Characters, conversational characters built on GWM-1, as able to explain concepts and answer questions; we found no named education customer.",
      "date": "2025-12",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://runway.com/research/introducing-runway-gwm-1",
        "title": "Introducing Runway GWM-1",
        "type": "blog",
        "date": "2025-12-11",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Odyssey",
      "what": "Odyssey says developers can use its world models for education and defense among other uses, and IQT (In-Q-Tel) took part in its June 2026 funding round; it names no training customer.",
      "date": "2026-06",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://odyssey.systems/investment-from-nvidia-and-samsung",
        "title": "Odyssey Announces Investment from NVentures and Samsung Next",
        "type": "blog",
        "date": "2026-02-12",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://odyssey.systems/our-series-b",
        "title": "Our $310 Million Fundraise to Accelerate World Simulation",
        "type": "blog",
        "date": "2026-06-17",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Training simulators need accurate places and physics, and Google DeepMind says Genie 3 cannot yet represent real-world locations with geographic accuracy and supports only a few minutes of continuous interaction.",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": []
   },
   {
    "id": "agent-training",
    "name": "Training AI agents in generated worlds",
    "plain": "A software agent, such as a game-playing AI, practises inside worlds that a world model generates instead of inside a hand-built game or simulator. This gives the agent an unlimited supply of new practice worlds and lets it learn from recorded video without touching the real environment.",
    "maturity": "research",
    "maturity_basis": "The evidence is research from Google DeepMind, NVIDIA and academic groups; Google DeepMind shares SIMA 2 only as a limited research preview, and no product for training agents in generated worlds was found.",
    "evidence": [
     {
      "org": "Google DeepMind",
      "what": "Google DeepMind tested its SIMA 2 game agent in new worlds generated by Genie 3 and reports that SIMA 2 can improve itself through trial and error with Gemini-based feedback; SIMA 2 is a limited research preview for a small group of academics and game developers.",
      "date": "2025-11",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/sima-2-an-agent-that-plays-reasons-and-learns-with-you-in-virtual-3d-worlds/",
        "title": "SIMA 2: An agent that plays, reasons, and learns with you in virtual 3D worlds",
        "type": "blog",
        "date": "2025-11-13",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "NVIDIA (GEAR team)",
      "what": "NVIDIA says its GEAR research team uses Cosmos 3 to develop video action models that help agents learn to reason, move and act across games, simulations and real robots.",
      "date": "2026-05",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://blogs.nvidia.com/blog/cosmos-3-physical-ai-open-world-foundation-model/",
        "title": "How Cosmos 3 Helps Physical AI Think Before It Acts",
        "type": "blog",
        "date": "2026-05-31",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Danijar Hafner, Wilson Yan, Timothy Lillicrap (Dreamer 4)",
      "what": "Dreamer 4 trains an agent by reinforcement learning inside a world model learned from recorded Minecraft play, and its authors report it is the first agent to obtain diamonds in Minecraft purely from offline data, a task needing over 20,000 mouse and keyboard actions.",
      "date": "2025-09",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2509.24527",
        "title": "Training Agents Inside of Scalable World Models",
        "type": "paper",
        "date": "2025-09-29",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Danijar Hafner and co-authors (DreamerV3)",
      "what": "DreamerV3, published in Nature in April 2025, learns a model of its environment and improves by imagining future scenarios; with one configuration it outperformed specialised methods on over 150 tasks and was the first algorithm to collect diamonds in Minecraft from scratch without human data.",
      "date": "2023-01",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2301.04104",
        "title": "Mastering Diverse Domains through World Models",
        "type": "paper",
        "date": "2023-01-10",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://www.nature.com/articles/s41586-025-08744-2",
        "title": "Mastering diverse control tasks through world models (Nature)",
        "type": "paper",
        "date": "2025-04-02",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "University of Geneva; University of Edinburgh; Microsoft Research (DIAMOND)",
      "what": "DIAMOND trains a reinforcement learning agent entirely inside a diffusion world model and reports a mean human-normalised score of 1.46 on the Atari 100k benchmark, a best for agents trained only in a world model at the time.",
      "date": "2024-05",
      "kind": "paper",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2405.12399",
        "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
        "type": "paper",
        "date": "2024-05-20",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://diamond-wm.github.io/",
        "title": "DIAMOND project page",
        "type": "site",
        "date": "2024-05",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Generated worlds still limit what an agent can do: Google DeepMind says the range of actions an agent can take directly in Genie 3 is constrained and that modelling several independent agents in one world is an open research problem.",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/",
        "title": "Genie 3: A new frontier for world models",
        "type": "blog",
        "date": "2025-08-05",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "text": "World models make mistakes about object interactions, and Dreamer 4's authors note that earlier world models could not predict object interactions in complex environments accurately, which limits what an agent can learn from them.",
      "level": "verified",
      "sources": [
       {
        "url": "https://arxiv.org/abs/2509.24527",
        "title": "Training Agents Inside of Scalable World Models",
        "type": "paper",
        "date": "2025-09-29",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "TechCrunch reported in June 2026 that General Intuition, a company spun out of the game-clip platform Medal to train agents on gameplay video, raised $320 million at a $2.3 billion valuation.",
      "level": "reported",
      "sources": [
       {
        "url": "https://techcrunch.com/2026/06/25/general-intuitions-2-3b-bet-that-video-games-can-train-ai-agents-for-the-real-world/",
        "title": "General Intuition's $2.3B bet that video games can train AI agents for the real world (TechCrunch)",
        "type": "secondary",
        "date": "2026-06-25",
        "accessed": "2026-10-11"
       }
      ]
     }
    ]
   },
   {
    "id": "healthcare-surgical",
    "name": "Surgical and medical robot simulation",
    "plain": "A world model trained on surgical video predicts how tissue and instruments will look and move after a robot's actions, or makes simulated surgical video look real. Surgical robot makers want this to train and test robot software without patients.",
    "maturity": "pilot",
    "maturity_basis": "NVIDIA names surgical robot makers (CMR Surgical, Johnson & Johnson MedTech, LEM Surgical, Moon Surgical) that use Cosmos-based tools in development, and it has released open surgical world models, but none reports clinical use.",
    "evidence": [
     {
      "org": "CMR Surgical; NVIDIA",
      "what": "NVIDIA says CMR Surgical is using Cosmos-H simulation to train and validate robotic intelligence for its Versius surgical system before clinical deployment.",
      "date": "2026-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-and-global-robotics-leaders-take-physical-ai-to-the-real-world",
        "title": "NVIDIA and Global Robotics Leaders Take Physical AI to the Real World",
        "type": "press-release",
        "date": "2026-03-16",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Johnson & Johnson MedTech; NVIDIA",
      "what": "NVIDIA says Johnson & Johnson MedTech is using Isaac Sim- and Cosmos-based post-training workflows to train and validate systems for its Monarch Platform for Urology.",
      "date": "2026-03",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-and-global-robotics-leaders-take-physical-ai-to-the-real-world",
        "title": "NVIDIA and Global Robotics Leaders Take Physical AI to the Real World",
        "type": "press-release",
        "date": "2026-03-16",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "LEM Surgical; NVIDIA",
      "what": "NVIDIA says LEM Surgical is using Isaac for Healthcare and Cosmos Transfer to train the autonomous arms of its Dynamis surgical robot.",
      "date": "2026-01",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-releases-new-physical-ai-models-as-global-partners-unveil-next-generation-robots",
        "title": "NVIDIA Releases New Physical AI Models as Global Partners Unveil Next-Generation Robots",
        "type": "press-release",
        "date": "2026-01-05",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "Moon Surgical; Virtual Incision; NVIDIA",
      "what": "NVIDIA lists Moon Surgical as using Cosmos Transfer (August 2025) and said Virtual Incision was exploring Cosmos for future surgical robots (March 2025).",
      "date": "2025-08",
      "kind": "customer",
      "level": "verified",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-opens-portals-to-world-of-robotics-with-new-omniverse-libraries-cosmos-physical-ai-models-and-ai-computing-infrastructure",
        "title": "NVIDIA Opens Portals to World of Robotics With New Omniverse Libraries, Cosmos Physical AI Models and AI Computing Infrastructure",
        "type": "press-release",
        "date": "2025-08-11",
        "accessed": "2026-10-10"
       },
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-announces-major-release-of-cosmos-world-foundation-models-and-physical-ai-data-tools",
        "title": "NVIDIA Announces Major Release of Cosmos World Foundation Models and Physical AI Data Tools",
        "type": "press-release",
        "date": "2025-03-18",
        "accessed": "2026-10-10"
       }
      ]
     },
     {
      "org": "NVIDIA",
      "what": "NVIDIA released Cosmos-H-Surgical-Simulator, a surgical world model driven by robot kinematics and built on Cosmos-Predict2.5-2B fine-tuned on the Open-H surgical benchmark, and Cosmos-H-Dreams, which streams surgical video live in response to a human operator's or a robot policy's actions; both are under NVIDIA's own licence terms.",
      "date": "2026-07",
      "kind": "demo",
      "level": "verified",
      "sources": [
       {
        "url": "https://huggingface.co/nvidia/Cosmos-H-Surgical-Simulator",
        "title": "nvidia/Cosmos-H-Surgical-Simulator (Hugging Face model card)",
        "type": "repo",
        "date": "2026-02",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://huggingface.co/nvidia/Cosmos-H-Dreams",
        "title": "nvidia/Cosmos-H-Dreams (Hugging Face model card)",
        "type": "repo",
        "date": "2026-07",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Surgical robots need regulatory clearance, and the named projects use world models to train and validate before clinical deployment; none reports clinical outcomes.",
      "level": "inferred",
      "sources": [
       {
        "url": "https://nvidianews.nvidia.com/news/nvidia-and-global-robotics-leaders-take-physical-ai-to-the-real-world",
        "title": "NVIDIA and Global Robotics Leaders Take Physical AI to the Real World",
        "type": "press-release",
        "date": "2026-03-16",
        "accessed": "2026-10-10"
       }
      ]
     }
    ],
    "market_signals": [
     {
      "what": "On 2026-10-10 the Hugging Face API reported all-time download counts of 80,129 for nvidia/Cosmos-H-Surgical, 1,853 for nvidia/Cosmos-H-Surgical-Simulator and 2,243 for nvidia/Cosmos-H-Dreams.",
      "level": "verified",
      "sources": [
       {
        "url": "https://huggingface.co/api/models?author=nvidia&search=Cosmos&sort=downloads&direction=-1&limit=40&expand[]=downloads&expand[]=downloadsAllTime&expand[]=createdAt&expand[]=likes",
        "title": "Hugging Face Hub API: NVIDIA Cosmos models with download counts",
        "type": "index",
        "date": "2026-10-10",
        "accessed": "2026-10-10"
       }
      ]
     }
    ]
   },
   {
    "id": "systems-control",
    "name": "Control of computer systems and industrial plants",
    "plain": "A learned model predicts how a system such as a video encoder or a cooling plant will respond to different control settings, and the controller picks the settings that the model expects to work best. This is the model-based reinforcement learning line of world models applied outside games and robots, to save energy or bandwidth.",
    "maturity": "pilot",
    "maturity_basis": "Google runs these systems in its own data centres and on part of YouTube's traffic; we did not confirm a product of this kind sold to other companies.",
    "evidence": [
     {
      "org": "Google DeepMind; YouTube",
      "what": "Google DeepMind applied MuZero, which plans with a learned model, to video compression rate control and says that since launching to production on a portion of YouTube's live traffic it has shown an average 4% bitrate reduction (internal use).",
      "date": "2022-02",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/blog/muzeros-first-step-from-research-into-the-real-world/",
        "title": "MuZero's first step from research into the real world",
        "type": "blog",
        "date": "2022-02-11",
        "accessed": "2026-10-11"
       },
       {
        "url": "https://arxiv.org/abs/2202.06626",
        "title": "MuZero with Self-competition for Rate Control in VP9 Video Compression",
        "type": "paper",
        "date": "2022-02-14",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Google DeepMind; Google data centres",
      "what": "Every five minutes Google's cooling control system feeds sensor snapshots into neural networks that predict how candidate actions will affect future energy use, then picks safe actions; DeepMind reported energy savings of around 30 percent on average in 2018 (internal use).",
      "date": "2018-08",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/discover/blog/safety-first-ai-for-autonomous-data-centre-cooling-and-industrial-control/",
        "title": "Safety-first AI for autonomous data centre cooling and industrial control",
        "type": "blog",
        "date": "2018-08-17",
        "accessed": "2026-10-11"
       }
      ]
     },
     {
      "org": "Google DeepMind; Google data centres",
      "what": "In 2016 DeepMind trained neural networks to predict data-centre power efficiency, temperature and pressure for the next hour, and reported that their recommendations cut the energy used for cooling by 40 percent (internal use).",
      "date": "2016-07",
      "kind": "pilot",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/discover/blog/deepmind-ai-reduces-google-data-centre-cooling-bill-by-40/",
        "title": "DeepMind AI Reduces Google Data Centre Cooling Bill by 40%",
        "type": "blog",
        "date": "2016-07-20",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "blockers": [
     {
      "text": "Safety limits how far such controllers can go: DeepMind's 2018 system runs under operator supervision and a safety-first operating regime, which DeepMind says trades some energy savings for reliability.",
      "level": "verified",
      "sources": [
       {
        "url": "https://deepmind.google/discover/blog/safety-first-ai-for-autonomous-data-centre-cooling-and-industrial-control/",
        "title": "Safety-first AI for autonomous data centre cooling and industrial control",
        "type": "blog",
        "date": "2018-08-17",
        "accessed": "2026-10-11"
       }
      ]
     }
    ],
    "market_signals": []
   }
  ],
  "notes": "Checked 2026-10-10 to 2026-10-11 (the local date changed during the work; each source records its own access date). Definitions used: kind 'paper' = research paper or technical report; 'demo' = public demo, research preview or open release with no stated customers; 'pilot' = a named organisation uses a world model in its own operations or a trial, including internal production use, while not selling it for this purpose (the 'what' sentence says 'internal use'); 'product' = sold or generally available to outside users; 'customer' = a named outside organisation using a vendor's world model, as stated by the vendor or the customer. Maturity: 'product' if something is sold or generally available for the use with named users; 'pilot' if named organisations use it in operations or trials; 'research' otherwise. Many 'customer' items come only from vendor announcements (especially NVIDIA's); the 'what' sentence says so. Results and figures are the organisations' own claims unless stated. Blockers are stored as objects {text, level, sources} rather than plain strings so that each carries sources and an evidence level, as the brief requires for every item. Source type 'press-release' is added to the schema-v0 list; press releases hosted on SEC EDGAR or PR Newswire are the companies' own releases. Undated pages carry date 'undated'; evidence with no findable date carries 'unknown'. Several vendors call ordinary video generators 'world models' (Runway, OpenAI) and carmakers use 'world model' as a product name (NIO, Huawei); entries say so where it matters. NVIDIA's 'Cosmos' brand also covers a vision-language model (Cosmos Reason) that is not a world model. Systems control (MuZero, data-centre cooling) is included as the model-based reinforcement learning line applied outside robots and games; it is at the edge of this section's scope. No analyst market-size figures were included: we did not open any analyst firm's own page. No totals were computed. Coverage limit: the shared web-search budget ran out partway through (after the robotics and driving sections). Games, film, 3D, AR/VR, industrial, education, agent-training, surgical and systems-control evidence was gathered by opening official pages, sitemaps and papers directly, so these sections may miss organisations that a full search would find; see leads_not_confirmed."
 },
 "evaluation": {
  "benchmarks": [
   {
    "name": "1X World Model Challenge",
    "kind": "challenge",
    "domains": [
     "robotics"
    ],
    "org": "1X Technologies (2025 phases with OpenDriveLab)",
    "date": "2024-06",
    "measures": "How well models predict future first-person frames of 1X's EVE humanoid from past frames and actions.",
    "scoring": "Compression challenge: temporally teacher-forced cross-entropy loss on held-out image tokens (2024 target: below 8.0). Sampling challenge: generated future frames compared with held-out real frames (PSNR in the 2025 phases). A planned Evaluation Challenge (rank N policies inside a world model) was announced as upcoming.",
    "licence": "Code: Apache-2.0 (GitHub 1xgpt); Data: raw video CC-BY-NC-SA-4.0, tokenized data Apache-2.0 (HF cards)",
    "url": "https://github.com/1x-technologies/1xgpt",
    "in_atlas": true,
    "atlas_id": "1x-world-model-challenge",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/1x-technologies/1xgpt",
      "title": "1x-technologies/1xgpt",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/1x-technologies/1xgpt/main/README.md",
      "title": "1xgpt README (challenge rules)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry '1x-world-model-challenge' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "EVA-Bench (Embodied Video Anticipation Benchmark)",
    "kind": "benchmark",
    "domains": [
     "robotics",
     "general-video"
    ],
    "org": "Hong Kong University of Science and Technology, Peking University (State Key Laboratory of Multimedia Information Processing)",
    "date": "2024-10-20",
    "measures": "Embodied video anticipation by world models that combine a vision-language model and a video generator, across four meta-tasks: Action-Description, How-To, Finish-Thinking and Next-Step, on real robots, simulated robots and egocentric human activities, with in-domain and out-of-distribution samples.",
    "scoring": "Language outputs: BLEU-1, METEOR, ROUGE-L, CIDEr, SPICE, a CLIP score normalised to 0-1, and a GPT-4o score. Video outputs: VBench Subject Consistency, Background Consistency and Motion Smoothness, FVD, and Goal Completion Estimation (DreamSim distance to an annotated goal frame, min-max normalised over EVA-Bench). Finish-Thinking also reports yes/no VQA accuracy.",
    "licence": "unknown (looked at: arXiv full text, which has no code or data link; demo site sites.google.com/view/icml-eva, which offers raw code marked 'for rebuttal demo only, unready for publish' with no licence)",
    "url": "https://arxiv.org/abs/2410.15461",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2410.15461",
      "title": "EVA: An Embodied World Model for Future Video Anticipation (arXiv abs; v1 2024-10-20, v2 2025-06-10)",
      "type": "paper",
      "date": "2024-10-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2410.15461v2",
      "title": "EVA full text (arXiv HTML v2), Section 5 and Appendix C on EVA-Bench",
      "type": "paper",
      "date": "2025-06-10",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://sites.google.com/view/icml-eva",
      "title": "EVA demo site (labelled 'ICML2025 Submission')",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 125 curated samples (from the 500K-QA-pair EVA-Instruct set built from RoboVQA, Ego-Exo4D and other sources). Results: EVA scores 62.63 on the GPT-4o metric for Action-Description; EVA-Gen reaches GCE 86.83 and FVD 177.28 on Finish-Thinking. Venue: the demo site labels it an 'ICML2025 Submission'; no acceptance is shown on primary pages. Leaderboard: none. Authors overlap with WoW-World-Eval (Xiaowei Chi, Chun-Kai Fan, Shanghang Zhang, Yike Guo)."
   },
   {
    "name": "WorldSimBench",
    "kind": "benchmark",
    "domains": [
     "robotics",
     "driving",
     "games"
    ],
    "org": "arXiv and project page: The Chinese University of Hong Kong, Shenzhen; Shanghai Artificial Intelligence Laboratory; Beihang University; The University of Hong Kong. PMLR version adds Sun Yat-sen University, University of Oxford and the Guangdong Key Laboratory of Big Data Analysis and Processing.",
    "date": "2024-10-23",
    "measures": "Video generation models used as world simulators for embodied agents, in three scenarios: an open-ended embodied environment (Minecraft via MineRL), autonomous driving (CARLA) and robot manipulation (CALVIN). It checks perceived visual quality per embodied dimension and whether generated videos can be turned into correct control signals.",
    "scoring": "Explicit Perceptual Evaluation: a Human Preference Evaluator (Flash-VStream VideoLLM fine-tuned with LoRA on the HF-Embodied dataset of 35,701 human-scored tuples across 3 scenarios and 20 dimensions) scores 5 videos per prompt for 5 prompts per dimension; scale 1-2 for Minecraft and 1-5 for driving and manipulation, normalised to 0-1 for reporting. Implicit Manipulative Evaluation: video-to-action models turn generated videos into control in closed loop; Minecraft uses travel distance, dig depth and items collected; CARLA uses 8 metrics (route completion, infraction score, driving score, vehicle, pedestrian and layout collisions, red-light violations, off-road infractions); CALVIN (train A, B, C, test D) uses average success over 20 trials.",
    "licence": "unknown (looked at: project page, whose Code and Dataset buttons have no links; GitHub repository search, which returns only the homepage repo IranQin/WorldSimBench.github.io with no licence; HF dataset search for WorldSimBench and HF-Embodied, no results)",
    "url": "https://iranqin.github.io/WorldSimBench.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2410.18072",
      "title": "WorldSimBench: Towards Video Generation Models as World Simulators (arXiv abs, 13 authors)",
      "type": "paper",
      "date": "2024-10-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2410.18072",
      "title": "WorldSimBench full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2024-10-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://proceedings.mlr.press/v267/qin25f.html",
      "title": "WorldSimBench, Proceedings of ICML 2025, PMLR 267:50338-50362 (12 authors)",
      "type": "paper",
      "date": "2025-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/qin25f/qin25f.pdf",
      "title": "WorldSimBench PMLR PDF (affiliation footnote, Table 3)",
      "type": "paper",
      "date": "2025-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://iranqin.github.io/WorldSimBench.github.io",
      "title": "WorldSimBench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://api.github.com/search/repositories?q=WorldSimBench",
      "title": "GitHub repository search for WorldSimBench (only the homepage repo IranQin/WorldSimBench.github.io, no licence)",
      "type": "index",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Models: 8 video generators (Open-Sora-Plan T2V and TI2V, Lavie, ModelScope, OpenSora, AnimateDiff, DynamiCrafter, EasyAnimate), all fine-tuned per scenario. Venue: ICML 2025, PMLR 267:50338-50362; the project page also lists 'CVPR 2025 @WorldModelBench Oral'. Author-list conflict: arXiv and the project page list 13 authors including Wanli Ouyang; the PMLR page lists 12 without him. Leaderboard: none found."
   },
   {
    "name": "WorldModelBench",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "robotics",
     "driving",
     "games"
    ],
    "org": "UC Berkeley; UC San Diego; NVIDIA; MIT",
    "date": "2025-02-28",
    "measures": "Whether text-to-video and image-to-video generators behave as world models in seven application domains: autonomous driving, robotics, human activities, industrial, natural scenes, simulation gaming and animation. It checks instruction following, physics adherence and commonsense (frame-wise and temporal quality).",
    "scoring": "Each video gets up to 10 points: instruction following 0-3 (four levels); physics adherence 0-5 (one binary point each for Newton's first law, mass conservation and solid mechanics, fluid mechanics, impenetrability, gravitation); commonsense 0-2 (frame-wise quality, temporal quality). Crowd voters (65 voters, 8,336 complete votes, 67K labels) scored 14 models. Automatic judge: a 2B vision-language model (VILA family, released as Efficient-Large-Model/vila-ewm-qwen2-1.5b) fine-tuned on the votes; it outputs the 8 per-criterion labels.",
    "licence": "Code: no licence (GitHub API license null; no LICENSE file in WorldModelBench-Team/WorldModelBench). Data: no licence tag on the Hugging Face dataset Efficient-Large-Model/worldmodelbench (the README says data moved to GitHub on 2025-07-28) and no licence tag on the judge model; the README says annotators avoided sources that forbid redistribution.",
    "url": "https://worldmodelbench-team.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2502.20694",
      "title": "WorldModelBench: Judging Video Generation Models As World Models (arXiv abstract page)",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2502.20694",
      "title": "WorldModelBench (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldmodelbench-team.github.io",
      "title": "WorldModelBench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldmodelbench-team.github.io/leaderboard_data.json",
      "title": "WorldModelBench leaderboard data file",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/WorldModelBench-Team/WorldModelBench",
      "title": "WorldModelBench repository README",
      "type": "repo",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://openreview.net/forum?id=a3hafrDzuA",
      "title": "WorldModelBench on OpenReview (NeurIPS 2025 Datasets and Benchmarks Track poster; read via api2.openreview.net search)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/Efficient-Large-Model/worldmodelbench",
      "title": "worldmodelbench dataset metadata",
      "type": "dataset-card",
      "date": "2024-12-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 350 image-and-text conditions (50 per domain), 56 subdomains; 14 models (5 closed, 9 open variants); WorldModelBench-Hard has 45 prompts. Human results: KLING top with 8.82 of 10; 61% of KLING videos fully complete the instructed task; image-to-video variants score below their text-to-video counterparts. Venue: OpenReview lists the NeurIPS 2025 Datasets and Benchmarks Track (poster); the repo README says it was an oral paper at the CVPR 2025 WorldModelBench workshop. Leaderboard: yes, on the project page (13 rows, all dated 2024-11-15; top KLING 9.10) and an EvalAI challenge (https://eval.ai/web/challenges/challenge-page/2179/overview, not opened). The abstract and project page say the judge has '8.6% higher average accuracy' than GPT-4o; the introduction says 9.9% lower error rate. Affiliation block lists the four organisations without a per-author mapping in the HTML."
   },
   {
    "name": "EWMBench",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "AgiBot; Shanghai Jiao Tong University; MMLab-CUHK; Harbin Institute of Technology (arXiv header)",
    "date": "2025-05",
    "measures": "How closely robot-manipulation videos from a video generator match real AgiBot World episodes in scene, motion and task meaning.",
    "scoring": "Scene consistency, three motion metrics and four semantic metrics computed against ground-truth episodes; summed overall score (max 8).",
    "licence": "Code and data: CC-BY-NC-SA-4.0 (README and HF dataset card)",
    "url": "https://arxiv.org/abs/2505.09694",
    "in_atlas": true,
    "atlas_id": "ewmbench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.09694",
      "title": "EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'ewmbench' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "AgiBot World Challenge, World Model track",
    "kind": "challenge",
    "domains": [
     "robotics"
    ],
    "org": "AgiBot (2025 with OpenDriveLab)",
    "date": "2025-05",
    "measures": "How well models predict robot head-camera video from actions on held-out AgiBot World episodes.",
    "scoring": "2026: mean of PSNR (clipped to 0-35, divided by 35), scene consistency and nDTW, using EWMBench metrics.",
    "licence": "Code and data: CC BY-NC-SA 4.0 (README and HF dataset cards)",
    "url": "https://agibot-world.com/challenge2026",
    "in_atlas": true,
    "atlas_id": "agibot-world-challenge-wm",
    "level": "verified",
    "sources": [
     {
      "url": "https://agibot-world-icra26wm.hf.space/",
      "title": "AgiBot World Challenge ICRA 2026 World Model track",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'agibot-world-challenge-wm' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "EnerVerse-AC (EVAC) as policy evaluator",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "AgiBot; Shanghai Jiao Tong University; MMLab, CUHK",
    "date": "2025-05-14",
    "measures": "Whether success rates of a Go-1 policy evaluated inside the EVAC action-conditioned world model match real-robot success rates across four retrieval tasks and across three training steps.",
    "scoring": "Binary success judged by three independent evaluators on real executions and on EVAC-generated videos; 40 trials per task; results shown as bar charts with no correlation statistic.",
    "licence": "Code and data: CC BY-NC-SA 4.0 (README 'License' section of github.com/AgibotTech/EnerVerse-AC; no LICENSE file; GitHub API reports none); Model weights: cc-by-nc-sa-4.0 (Hugging Face agibot-world/EnerVerse-AC)",
    "url": "https://arxiv.org/abs/2505.09723",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.09723",
      "title": "EnerVerse-AC: Envisioning Embodied Environments with Action Condition (arXiv abstract, v1)",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.09723",
      "title": "EnerVerse-AC full text (arXiv HTML v1), Sec. 3 'Evaluator for Policy Model', Sec. 4.3, Appendix A.4.1",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.09723v1/Fig5_Comp_Real_CAE.svg",
      "title": "EnerVerse-AC Fig. 7: success rate per task and per learning step, real robot vs EVAC",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/AgibotTech/EnerVerse-AC",
      "title": "AgibotTech/EnerVerse-AC repository README (License section; no LICENSE file)",
      "type": "repo",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/models/agibot-world/EnerVerse-AC",
      "title": "Hugging Face API: agibot-world/EnerVerse-AC model (licence cc-by-nc-sa-4.0)",
      "type": "dataset-card",
      "date": "2025-05-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://annaj2178.github.io/EnerverseAC.github.io/",
      "title": "EnerVerse-AC project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "The README says the released weights were trained only on open-source AgiBot World data and exclude the failure trajectories used in the paper, because of commercial restrictions. A qualitative LIBERO comparison is in the appendix. Robot model not named in the evaluation section."
   },
   {
    "name": "DreamGen Bench",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "NVIDIA (GEAR lab, lead), University of Washington, KAIST, UCLA, UCSD, Caltech, NTU, University of Maryland, UT Austin",
    "date": "2025-05-19",
    "measures": "How well image-to-video world models adapt to a target robot embodiment and generalise to unseen objects, behaviours and environments: whether generated robot videos follow the instruction and obey physics. Setups: RoboCasa (simulated Franka) and three real Fourier GR1 humanoid splits (Object, Behavior, Environment).",
    "scoring": "Instruction following (IF): a VLM gives each generated video a binary 0/1 score for completing the instruction; Table 2 reports GPT-4o, Qwen2.5-VL and human (Hu) columns, as percentages. Physics alignment (PA): average of a VideoCon-Physics score (0-1) and a Qwen2.5-VL-7B binary physics judgment. The DreamGen Bench score used against downstream results is the average of IF (GPT-4o) and PA.",
    "licence": "Code: Apache-2.0 (LICENSE of github.com/NVIDIA/GR00T-Dreams, which contains the dreamgenbench evaluation scripts). Data: CC-BY-4.0 for the evaluation initial frames (HF card nvidia/EVAL-175, now resolving to nvidia/PhysicalAI-Robotics-GR00T-Eval) and for the GR1 training set (nvidia/GR1-100, resolving to nvidia/PhysicalAI-Robotics-GR00T-GR1).",
    "url": "https://research.nvidia.com/labs/gear/dreamgen",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.12705",
      "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models (arXiv abs)",
      "type": "paper",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.12705v2",
      "title": "DreamGen full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://research.nvidia.com/labs/gear/dreamgen",
      "title": "DreamGen project page (NVIDIA GEAR)",
      "type": "project-page",
      "date": "2025-05-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/NVIDIA/GR00T-Dreams/main/README.md",
      "title": "NVIDIA/GR00T-Dreams README, section 5 DreamGen Bench",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.github.com/repos/NVIDIA/GR00T-Dreams/license",
      "title": "GitHub licence record for NVIDIA/GR00T-Dreams (Apache License 2.0)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/nvidia-cosmos/cosmos-predict2/main/documentations/post-training_video2world_gr00t.md",
      "title": "Cosmos-Predict2 doc: Video2World post-training for DreamGen Bench (dataset links)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/nvidia/EVAL-175/resolve/main/README.md",
      "title": "HF dataset card nvidia/EVAL-175 (resolves to nvidia/PhysicalAI-Robotics-GR00T-Eval; CC-BY-4.0)",
      "type": "dataset-card",
      "date": "2025-06-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale (Table 2): RoboCasa 1,200 training trajectories and 48 evaluation frames; GR1 100 training trajectories and 50 (Object) + 47 (Behavior) + 30 (Environment) evaluation frames. The EVAL-175 dataset card says 123 initial frames, versus 127 GR1 frames in Table 2 (conflict). Models: Hunyuan, CogVideoX, WAN 2.1 and Cosmos, each zero-shot and fine-tuned (8 variants). Results: zero-shot models score near 0 on IF; fine-tuned Cosmos leads RoboCasa (IF-GPT 79.2, PA 61.5) and GR1-Object (IF-GPT 90.0, PA 73.0); fine-tuned WAN2.1 leads GR1-Behavior IF-GPT (72.3); GR1-Environment: Cosmos-sft IF-GPT 69.0, WAN2.1-sft PA 66.5. Conflict: the main text names Qwen-VL-2.5 as the IF judge, while the Figure 6 score uses IF (GPT). Leaderboard: none found on the project page or repository. The README notes the benchmark uses about 50 videos per dataset and a small open VLM, so it may not generalise to multi-view or detailed physics judgments. Venue: none shown."
   },
   {
    "name": "WorldEval",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Midea Group; East China Normal University",
    "date": "2025-05-25",
    "measures": "Whether success rates of real-robot manipulation policies, obtained by rolling each policy out inside an action-conditioned video world model, track and rank the same policies' real-robot success rates.",
    "scoring": "Latent action embeddings taken from each policy network (Policy2Vec) condition a Wan 2.1 14B image-to-video model fine-tuned with LoRA, starting from the real first frame and the instruction. Gemini-2.0 answers a yes/no success question for each generated video. Per-policy success rates on each task are compared with real-robot success rates using Pearson r and MMRV. FID of generated videos is offered as a cheaper proxy for simple tasks.",
    "licence": "Code: MIT for modifications by Yaxuan Li (LICENSE at repo root) plus Apache-2.0 for code taken from DiffSynth Studio (apache-original/LICENSE), as stated in the README of github.com/liyaxuanliyaxuan/Worldeval; Data: unknown (no dataset release found in the README or on worldeval.github.io); policy weights are linked on Hugging Face (kuromivv/pi0, kuromivv/DexVLA, kuromivv/diffusion_policy), licences not checked.",
    "url": "https://worldeval.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.19017",
      "title": "WorldEval: World Model as Real-World Robot Policies Evaluator (arXiv abs; v1 is the only version)",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Sections 3-4, Appendices A-C",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldeval.github.io",
      "title": "WorldEval project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/liyaxuanliyaxuan/Worldeval",
      "title": "WorldEval repository (LICENSE, README, apache-original/LICENSE)",
      "type": "repo",
      "date": "2025-05-19",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Robot: AgileX bimanual ALOHA-style system with two 6-DoF arms and one top RealSense camera. Five tasks (Bussing Table, Collect Toy, Place Cup, Handover Block, Strike Block); the paper states 40 real rollouts per task and over 1,000 real-world trials in total. Policies: Diffusion Policy, OpenVLA, DexVLA, pi0. The world model was trained on 1,400 real trajectories (8 H800 GPUs, about 11 hours). The paper does not define MMRV itself; it refers readers to SIMPLER (Li et al., 2024, arXiv 2405.05941). No venue found (OpenReview lists only the CoRR record)."
   },
   {
    "name": "WorldGym",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Stanford University; NYU; Google DeepMind",
    "date": "2025-05-31",
    "measures": "Whether success rates of VLA policies, obtained by Monte Carlo rollouts in an autoregressive action-conditioned video world model started from real first frames, match real-robot success rates and keep policy rankings. Also used to test policies on edited out-of-distribution scenes and instructions.",
    "scoring": "A latent diffusion transformer trained with Diffusion Forcing on Open X-Embodiment data predicts one frame per action in each policy action chunk. GPT-4o reads the rollout frames and the instruction and gives 0 or 1 per rollout (0, 0.5 or 1 where the task has a partial-credit criterion). Per-task success rates are averaged and compared with real-world rates using Pearson r.",
    "licence": "Code: no licence file (GitHub API reports no licence; repository root has no LICENSE) at github.com/world-model-eval/world-model-eval; the world-model checkpoint is shared through a Google Drive link with no licence stated; Data: trained on Open X-Embodiment and Bridge V2 (their licences not checked).",
    "url": "https://world-model-eval.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.00613",
      "title": "WorldGym: World Model as An Environment for Policy Evaluation (arXiv abs; v1 2025-05-31, v2 2025-09-29, v3 2025-09-30)",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym full text (arXiv HTML v3): Sections 3-4, Appendices B and E",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.00613v1",
      "title": "Evaluating Robot Policies in a World Model (WorldGym arXiv v1 full text)",
      "type": "paper",
      "date": "2025-05-31",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://world-model-eval.github.io/abstract.html",
      "title": "WorldGym project page (abstract page)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/world-model-eval/world-model-eval",
      "title": "WorldGym repository (README, root file listing)",
      "type": "repo",
      "date": "2025-06-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://openreview.net/forum?id=hidBHy1CAw",
      "title": "WorldGym on OpenReview, listed as ICLR 2026 Poster (checked through the api2.openreview.net notes search)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Venue: ICLR 2026 poster. arXiv v1 (2025-05-31) was titled 'Evaluating Robot Policies in a World Model', called the method WPE, and compared world-model policy values with the LIBERO simulator; it had no real-robot success-rate comparison. The real-robot comparison is in v3 (v2 not checked). Real-robot comparison uses the OpenVLA Bridge (WidowX) suite: 17 tasks, 3 policies, 10 trials per task per policy. The paper states all reported rollouts complete in under an hour on a single GPU."
   },
   {
    "name": "Real-robot goal-image planning test (V-JEPA 2-AC)",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "FAIR at Meta (V-JEPA 2 paper)",
    "date": "2025-06",
    "measures": "Whether an action-conditioned world model can drive a real robot by planning: the robot is given goal images and the world model is used to search for actions, with no task-specific training or reward.",
    "scoring": "Success rate out of 10 trials per task on two Franka arms in different labs (reach, grasp, reach with object, pick-and-place; cup and box objects), compared with Octo and with planning in NVIDIA's Cosmos action-conditioned video model.",
    "licence": "Not a released benchmark; the evaluation setups are lab-specific (no protocol code found for the robot tests on the pages read).",
    "url": "https://arxiv.org/abs/2506.09985",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.09985v1",
      "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Measures usefulness directly instead of agreement. Results (Table 2): V-JEPA 2-AC average 100% reach, 65%/25% grasp (cup/box), 80%/65% pick-and-place; Octo 15%/0% grasp, 15%/10% pick-and-place. Table 3 (Lab 2): planning with Cosmos took 4 minutes per action and reached 80% reach, 0% cup grasp and 0% pick-and-place; V-JEPA 2-AC took 16 seconds per action. All runs by Meta."
   },
   {
    "name": "1X World Model evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "1X Technologies",
    "date": "2025-06-16",
    "measures": "How often the world model's success/failure prediction matches real outcomes, and whether its scores pick the same checkpoints and architectures as real double-blind A/B evaluations on humanoid robots.",
    "scoring": "Alignment is the accuracy of the model's success/failure (state-value) predictions on held-out episodes. On the Arcade task each checkpoint gets the score (successes - 3 x resets + 0.3 x attempts) / attempts, compared with real evaluations in plots.",
    "licence": "Code: none released; Data: none released (looked at the blog post and the PDF report).",
    "url": "https://www.1x.tech/discover/redwood-ai-world-model",
    "in_atlas": true,
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/discover/redwood-ai-world-model",
      "title": "1X World Model (1X blog)",
      "type": "blog",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms (technical progress report)",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Atlas entry 1x-world-model-eval. Newer 1X posts checked: 2026-01-12 (1XWM used as the robot policy) and 2026-06-04 (World Model Lab announcement); neither reports new policy-evaluation agreement numbers."
   },
   {
    "name": "WM-ABench",
    "kind": "benchmark",
    "domains": [
     "research",
     "robotics",
     "driving"
    ],
    "org": "Maitrix.org, UC San Diego, Johns Hopkins University, Cornell Tech, EPFL, University of Michigan",
    "date": "2025-06-27",
    "measures": "Whether vision-language models (not video generators) behave as internal world models, using atomic tests of perception (visual, spatial, temporal, quantitative, motion) and prediction (mechanistic simulation, transitive inference, compositional inference), with controlled counterfactual simulations.",
    "scoring": "Accuracy per fine-grained dimension (23 dimensions) on questions built in simulated environments (appendix sections cover ThreeDWorld, ManiSkill, Physion, CARLA and Habitat; the paper says 6 environments), compared with random chance and with measured human accuracy.",
    "licence": "Data: Apache-2.0 (HF dataset card maitrix-org/WM-ABench). Code: unknown (no code repository linked from the paper or project page).",
    "url": "https://wm-abench.maitrix.org/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.21876",
      "title": "Do Vision-Language Models Have Internal World Models? Towards an Atomic Evaluation (arXiv abs; comment: ACL 2025 (Findings))",
      "type": "paper",
      "date": "2025-06-27",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2506.21876",
      "title": "WM-ABench full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-06-27",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wm-abench.maitrix.org/",
      "title": "WM-ABench project page with per-category leaderboard",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/maitrix-org/WM-ABench/resolve/main/README.md",
      "title": "HF dataset card maitrix-org/WM-ABench (license: apache-2.0)",
      "type": "dataset-card",
      "date": "2025-08-29",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 660 experiments on 15 commercial and open-source VLMs; HF size category 100K-1M items. Example results: best model 43.8% on multi-step navigation in Habitat vs human 90.0%; best 40.2% on TDW collision prediction and 51.3% on ManiSkill manipulation vs human 84.0% and 88.0%; almost all models near random on distinguishing motion trajectories. Venue: ACL 2025 (Findings), per the arXiv comment; the project page says ACL 2025. Leaderboard: yes, per-category tables on the project page (no single overall ranking; rows are not sorted by score). This benchmark targets VLMs as world models, which differs from the video world models in the other rows."
   },
   {
    "name": "IRASim policy evaluation (LIBERO)",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Hong Kong University of Science and Technology; ByteDance Seed",
    "date": "2025-07-29",
    "measures": "Whether success rates of diffusion-policy checkpoints judged in IRASim-generated rollouts match the success rates measured in the LIBERO MuJoCo simulator.",
    "scoring": "Pearson correlation between the two sets of success rates; 4 checkpoints, 50 runs each in IRASim and in the simulator; humans judge success in IRASim rollouts.",
    "licence": "Code: Apache-2.0 (github.com/bytedance/IRASim, GitHub API); Data and checkpoints: licence not found in the places checked (repo metadata only)",
    "url": "https://arxiv.org/abs/2406.14540",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2406.14540",
      "title": "IRASim: A Fine-Grained World Model for Robot Manipulation (arXiv abstract; v1 2024-06-20, v2 2025-07-29)",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2406.14540",
      "title": "IRASim full text (arXiv HTML v2), Sec. 4.2 and Table 4",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2406.14540v1",
      "title": "IRASim arXiv HTML v1 (no policy-evaluation section, no Pearson value)",
      "type": "paper",
      "date": "2024-06-20",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/bytedance/IRASim",
      "title": "bytedance/IRASim repository (GitHub API licence: Apache-2.0)",
      "type": "repo",
      "date": "2024-06-19",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Date is arXiv v2, which added the policy-evaluation section; v1 (2024-06-20) has none. Comparison target is a simulator. IRASim is a trajectory-conditioned video diffusion transformer, initialised from OpenSora for this experiment and trained on expert demonstrations plus policy rollouts with failures."
   },
   {
    "name": "PAI-Bench (Physical AI Bench)",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "robotics",
     "driving"
    ],
    "org": "Georgia Tech; Carnegie Mellon University (NVIDIA acknowledged for support)",
    "date": "2025-09",
    "measures": "Video generation and video understanding on physical-AI clips in six domains: common sense, driving, robots, industry, people, physics.",
    "scoring": "Generation: Overall = 0.5 x Quality score (frame consistency, motion smoothness, aesthetics, video-text alignment) + 0.5 x Domain score (VLM judge answers yes/no questions). Understanding: multiple-choice accuracy.",
    "licence": "Code: MIT (LICENSE file); Data: PAI-Bench-G CC-BY-NC-4.0, PAI-Bench-C and -U MIT (HF cards)",
    "url": "https://github.com/SHI-Labs/physical-ai-bench",
    "in_atlas": true,
    "atlas_id": "pai-bench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.01989v1",
      "title": "PAI-Bench: A Comprehensive Benchmark For Physical AI",
      "type": "paper",
      "date": "2025-12-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'pai-bench' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "Ctrl-World",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Stanford University; Tsinghua University",
    "date": "2025-10-11",
    "measures": "Whether the instruction-following rate and the task success rate of generalist DROID policies, rolled out closed-loop inside a multi-view action-conditioned video world model, match the same policies' rates on a real robot in a new DROID setup.",
    "scoring": "Real and world-model rollouts start from the same initial observations. Human annotators label each trajectory twice against written per-task criteria: whether it follows the instruction and whether it fully succeeds. The paper plots world-model rates against real rates for each policy-task pair and reports linear fits; it reports no correlation coefficient.",
    "licence": "Code: MIT (LICENSE.txt, copyright 2025 Tsinghua University, github.com/Robert-gyj/Ctrl-World); Weights: Hugging Face card yjguo/Ctrl-World declares MIT and lists stabilityai/stable-video-diffusion-img2vid as base model, whose card declares the stable-video-diffusion-community licence; Data: trained on the public DROID dataset (licence not checked).",
    "url": "https://ctrl-world.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abs; v1 2025-10-11, v3 2026-03-01)",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.10125",
      "title": "Ctrl-World full text (arXiv HTML v3): Section 5.3, Appendix B Table 3",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2510.10125v3",
      "title": "Ctrl-World PDF v3 (Figure 7 on page 8; ICLR 2026 header)",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://ctrl-world.github.io",
      "title": "Ctrl-World project page (lists ICLR 2026)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Robert-gyj/Ctrl-World",
      "title": "Ctrl-World repository (LICENSE.txt, readme.md)",
      "type": "repo",
      "date": "2025-10-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/models/yjguo/Ctrl-World",
      "title": "Ctrl-World model card metadata (license: mit)",
      "type": "repo",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/models/stabilityai/stable-video-diffusion-img2vid",
      "title": "Stable Video Diffusion img2vid model card metadata (license_name: stable-video-diffusion-community)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://openreview.net/forum?id=748bHL2BAv",
      "title": "Ctrl-World on OpenReview, listed as ICLR 2026 Poster (checked through the api2.openreview.net notes search)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Venue: ICLR 2026. The evaluation section and Table 3 are the same in arXiv v1 and v3. World model trained on DROID (95,599 trajectories, 564 scenes) and used zero-shot on the authors' own DROID platform (Franka Panda arm, Robotiq gripper, one wrist camera and two third-person cameras). Policies: pi0-droid, pi0-FAST-droid, pi0.5-droid; 7 tasks (Pick-Place, Fold-Towel, Drawer, Wipe-table, Close-laptop, Pull-tissue, Stack). The repository README states each task category was run 20 times in the paper. The paper also uses generated rollouts as fine-tuning data, which raised pi0.5 success on new instructions from 38.7% to 83.4%."
   },
   {
    "name": "Cosmos-Surg-dVRK",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "NVIDIA; Johns Hopkins University; Stanford University",
    "date": "2025-10-17",
    "measures": "Whether success rates of surgical robot policies, rolled out online in a fine-tuned Cosmos world foundation model, match success rates of the same policies on a real da Vinci Research Kit (dVRK Si).",
    "scoring": "Cosmos-Predict2-2B-Video2World is fine-tuned on dVRK endoscope video paired with kinematics. Each world-model rollout starts from the first endoscope frame of a recorded real trial and is generated with 3 seeds. Success is labelled by two human raters or by a V-JEPA 2 attentive-probe classifier. Agreement is reported as Pearson r, MMRV, mean bias error and ICC.",
    "licence": "Code: no release found (looked at the arXiv paper and abs page); Weights: no release found (same places); the base model Cosmos-Predict2 is cited in the paper as Apache-2.0 code available under the NVIDIA Open Model License; Data: tabletop dataset from Haworth et al. 2025 and cholecystectomy dataset from Kim et al. 2025 (licences not checked).",
    "url": "https://arxiv.org/abs/2510.16240",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.16240",
      "title": "Cosmos-Surg-dVRK: World Foundation Model-based Automated Online Evaluation of Surgical Robot Policy Learning (arXiv abs; v1 2025-10-17, v2 2025-11-03)",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Sections 3-6, Tables 1-3",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.crossref.org/works/10.1109/LRA.2026.3675962",
      "title": "Crossref record: IEEE Robotics and Automation Letters 11(5): 5978-5985",
      "type": "index",
      "date": "2026-05",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Published in IEEE Robotics and Automation Letters, vol. 11, no. 5, pp. 5978-5985, May 2026 (DOI 10.1109/LRA.2026.3675962). Numbers in this fragment come from arXiv v2; the RA-L text was not read. Tabletop suture-pad tasks: needle pickup, needle handover, needle throw, knot tying. Ex-vivo porcine cholecystectomy tasks: apply first clip, apply third clip, cut cystic duct."
   },
   {
    "name": "World-in-World",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Johns Hopkins University (lead; corresponding author Jieneng Chen), Peking University, Princeton University, MIT, Harvard University",
    "date": "2025-10-20",
    "measures": "Whether visual world models help an embodied agent succeed in closed-loop tasks: Active Recognition (AR), Image-Goal Navigation (ImageNav), Active Embodied Question Answering (A-EQA) and robotic manipulation. Each world model is plugged into the same proposal-simulation-revision planning loop through a unified action API (text prompt, camera trajectory or low-level actions).",
    "scoring": "Task success is the primary metric: AR success rate and mean trajectory length; ImageNav success rate, mean trajectory length and SPL; A-EQA answer score, mean trajectory length and SPL; manipulation success rate and mean trajectory length. Generation quality is reported separately as the average of an aesthetic predictor and a MUSIQ image-quality predictor; controllability as 1 - LPIPS between predicted and ground-truth observations. Proposal policies include a heuristic policy, a 72B VLM and a 3D diffusion policy.",
    "licence": "Code: MIT (LICENSE in github.com/World-In-World/world-in-world). Data: the evaluation-episode dataset on HF (zonszer/WIW_datasets) has no licence field; scene datasets (Matterport3D for AR, HM3D for ImageNav and A-EQA) are downloaded separately as described in the README.",
    "url": "https://world-in-world.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2510.18135",
      "title": "World-in-World: World Models in a Closed-Loop World (arXiv abs; v1 2025-10-20, v2 2026-08-16)",
      "type": "paper",
      "date": "2025-10-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.18135v2",
      "title": "World-in-World full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-08-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-in-world.github.io/",
      "title": "World-in-World project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-in-world.github.io/subpages/leaderboard.html",
      "title": "World-In-World Leaderboard",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/World-In-World/world-in-world/main/README.md",
      "title": "World-in-World GitHub README",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.github.com/repos/World-In-World/world-in-world/license",
      "title": "GitHub licence record for World-In-World/world-in-world (MIT License, LICENSE)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/zonszer/WIW_datasets",
      "title": "HF dataset zonszer/WIW_datasets metadata (no license field)",
      "type": "dataset-card",
      "date": "2025-10-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: AR 551 episodes in 29 Matterport3D scenes (Habitat-Sim); ImageNav 144 episodes from 87 HM3D scenes; A-EQA 184 questions in 54 scenes (OpenEQA split, HM3D); manipulation 4 RLBench tasks x 50 episodes (CoppeliaSim) in the paper; the repository added OpenPI as a proposer (2026-04-01) and LIBERO as a manipulation backend (2026-04-03). Models: PathDreamer, SE3DS, NWM, SVD, LTX-Video, Hunyuan, Wan2.1, Wan2.2 (5B and A14B), Cosmos-Predict2, Runway Gen4, plus post-trained variants. Venue: ICLR 2026 Oral (arXiv comment and project page). v2 (2026-08-16) only adds acknowledgements per the arXiv comment. Leaderboard: yes, https://world-in-world.github.io/subpages/leaderboard.html, 16 entries ranked by AR success; top Runway Gen4 64.79%, then Wan2.1 post-trained 62.61% and SVD post-trained 60.98%; new results are submitted by GitHub pull request. GitHub stars: 211 (2026-10-10)."
   },
   {
    "name": "Scalable Policy Evaluation with Video World Models (NVIDIA)",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "NVIDIA Research; University of Toronto; Vector Institute",
    "date": "2025-11-14",
    "measures": "Whether success rates predicted by action-conditioned video models (post-trained Cosmos-Predict2-2B) with a VLM success judge match simulator success rates on four RoboMimic tasks and real success rates of three generalist policies on four Bridge-setup tasks.",
    "scoring": "Pearson correlation and MMRV between predicted and actual success rates.",
    "licence": "unknown; the paper says the pipeline will be released (looked at: arXiv paper)",
    "url": "https://arxiv.org/abs/2511.11520",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models (arXiv abstract; v1 2025-11-14, v3 2025-12-04)",
      "type": "paper",
      "date": "2025-11-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520v3/real_world_video_model_correlation_plot_icra.svg",
      "title": "Tseng et al. Fig. 6: policy evaluation on the Bridge setup (Cosmos and IRASim, MMRV and Pearson)",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Real-world part follows SIMPLER: OCTO-Small, OCTO-Base and OpenVLA on 'lift the pot/carrot/eggplant/cup'; the video model is trained on Bridge V2 plus 300 trajectories collected in the authors' replicated setup. The appendix with real-world details is not in the arXiv HTML or PDF."
   },
   {
    "name": "Veo (Robotics) policy evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Google DeepMind (Gemini Robotics Team)",
    "date": "2025-12-11",
    "measures": "Whether success rates of Gemini Robotics On-Device policy checkpoints, predicted by closed-loop rollouts in a Veo-based action-conditioned multi-view video model, match real ALOHA 2 success rates in nominal scenes and in edited out-of-distribution scenes, and whether the model finds unsafe behaviour that also occurs on the real robot.",
    "scoring": "Veo 2 fine-tuned on robot data generates 8-second rollouts in four tiled camera views, conditioned on the first frames, the instruction and future robot poses. Out-of-distribution scenes are made by editing the overhead image with Gemini 2.5 Flash Image and filling the other views with a multi-view Veo 2 model. Human evaluators score each generated rollout as success or failure. Agreement with real success rates is reported as Pearson coefficient and MMRV.",
    "licence": "Code: none released (looked at the paper and veo-robotics.github.io); Data: none released (same places). The arXiv paper text is CC BY 4.0.",
    "url": "https://veo-robotics.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.10675",
      "title": "Evaluating Gemini Robotics Policies in a Veo World Simulator (arXiv abs; v1 2025-12-11, v2 2026-01-06)",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Sections 2-5 and 7",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://veo-robotics.github.io",
      "title": "Veo Robotics project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Technical report with no venue. Real-robot trials were run by the same team: 1600+ real-world evaluations of eight policy checkpoints on five tasks. The nominal test set has 80 scene-instruction combinations, including rephrased, misspelled, other-language and more or less specific instructions."
   },
   {
    "name": "PolaRiS",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "University of Washington; Princeton University; UC Berkeley; Stanford University; Toyota Research Institute; University of Southern California; Cornell University; Physical Intelligence",
    "date": "2025-12-18",
    "measures": "Whether policy scores in simulated copies of real scenes, built from short video scans with 2D Gaussian splatting, match real-robot scores of generalist DROID policies.",
    "scoring": "Each policy is co-fine-tuned for 1k steps on about 350 simulated demonstrations collected in 15 separate scenes, then run 50 times per task in simulation. Simulated rollouts are scored automatically on a 0-1 progress scale from simulator state. Real rollouts (20 per policy per environment) are graded by a human with the same rubric. Agreement is reported as Pearson r and MMRV.",
    "licence": "Code: MIT (LICENSE, copyright 2025 Arhan Jain, github.com/arhanjain/polaris); Data: PolaRiS-Hub environments on Hugging Face (owhan/PolaRiS-Hub), dataset card licence MIT.",
    "url": "https://polaris-evals.github.io",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.16881",
      "title": "PolaRiS: Scalable Real-to-Sim Evaluations for Generalist Robot Policies (arXiv abs; v1 2025-12-18, v2 2025-12-30)",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS full text (arXiv HTML v2): Sections 3, 5, Appendices C-D",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://polaris-evals.github.io",
      "title": "PolaRiS project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/arhanjain/polaris",
      "title": "PolaRiS repository (LICENSE)",
      "type": "repo",
      "date": "2025-11-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/owhan/PolaRiS-Hub",
      "title": "PolaRiS-Hub dataset card metadata (license: mit)",
      "type": "dataset-card",
      "date": "2026-03-14",
      "accessed": "2026-10-10"
     }
    ],
    "note": "PolaRiS is a physics simulator with Gaussian-splat rendering. It is not a learned world model. It is listed here because its paper measures the Ctrl-World video world model as a baseline evaluator and because Runway compares GWM-Robotics against it. Robot: DROID platform (7-DoF Franka Panda, two ZED cameras, one on the wrist). Six paired real and simulated environments at UW and Princeton. No venue found on OpenReview."
   },
   {
    "name": "RBench",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Peking University; ByteDance Seed",
    "date": "2026-01",
    "measures": "Whether video generators produce correct and physically plausible robot task videos across five task types and four robot body types.",
    "scoring": "MLLM judges and vision operators score task completion (physical-semantic plausibility, task adherence) and visual quality; final score is the mean over 650 cases.",
    "licence": "Code: unknown (no LICENSE file); Data: CC-BY-4.0 (HF dataset card)",
    "url": "https://arxiv.org/abs/2601.15282",
    "in_atlas": true,
    "atlas_id": "rbench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.15282v1",
      "title": "Rethinking Video Generation Model for the Embodied World (RBench)",
      "type": "paper",
      "date": "2026-01-21",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'rbench' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "WoW-World-Eval (Wow, wo, val)",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Peking University (State Key Laboratory of Multimedia Information Processing, School of Computer Science), Beijing Innovation Center of Humanoid Robotics, The Hong Kong University of Science and Technology",
    "date": "2026-01-07",
    "measures": "Image-plus-text-to-video generation for robot manipulation, framed as an 'Embodied Turing Test' over five abilities: perception, planning, prediction, generalisation and execution. Includes a human Turing test (can people tell generated from real video) and an inverse-dynamics-model (IDM) Turing test (can actions recovered from generated video be executed on a real robot).",
    "scoring": "22 metrics in five groups: video quality (FVD, PSNR, SSIM, DINO, DreamSim); instruction understanding judged by GPT-4o (Caption Score 1-5, Sequence Match 0-1, Execution Quality 1-5); physical law (mask-guided regional consistency with GroundedSAM2 and DINOv3, SAM2-tracked robot and object trajectories scored by L2, DTW and Frechet distance, a physical-common-sense score from Qwen-2.5-VL fine-tuned with GRPO on 1,297 human ratings, camera ATE and RPE); planning (DAG-based node correctness plus task completion, (sum) x 50); execution (real-robot success of actions from the GC-IDM). Each metric is clipped to fixed anchors, scaled to [0,1], passed through a monotone transform and mapped to (0,100); group scores are means; the overall score is a weighted mean of groups (equal weights give an unweighted mean).",
    "licence": "unknown (looked at: arXiv abs and full text, which give no code or data link; GitHub repository search and HF dataset search for WoW-World-Eval and wow-wo-val, no matches)",
    "url": "https://arxiv.org/abs/2601.04137",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.04137",
      "title": "Wow, wo, val! A Comprehensive Embodied World Model Evaluation Turing Test (arXiv abs)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2601.04137v1/correlation.png",
      "title": "WoW-World-Eval Figure 3b: overall score on metric vs overall score on human (r = 0.93, rho = 0.91)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 609 samples (RoboMIND, DROID, in-house robot data and AI-edited out-of-distribution images); 25 long-horizon planning samples; 9 real-world IDM tasks. Models: Kling 2.1, Hailuo I2V-02, CogVideoX1.5-I2V-5B, Wan2.1-I2V-14B, Cosmos-Predict1-7B, Cosmos-Predict2-2B and three WoW variants (the authors' own model, Chi et al.). Results: best overall Hailuo 52.55, best open-source WoW-cosmos2 50.74; best planning 17.27 (Hailuo); best physical-law score 68.02 (Kling, whose overall is 37.93). Human studies: 15 domain experts rated over 1,200 real and generated videos on four 1-5 dimensions; 13 participants took the human Turing test. Leaderboard: none found. Venue: none shown."
   },
   {
    "name": "DreamDojo policy evaluation (AgiBot fruit packing)",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "NVIDIA (lead); co-author affiliations also list HKUST, UC Berkeley, KAIST, University of Toronto, UC San Diego, University of Washington, Stanford, UT Austin",
    "date": "2026-02-06",
    "measures": "Whether the success rates of policy checkpoints simulated inside the DreamDojo world model match the success rates of the same checkpoints on a real robot, on one long-horizon fruit-packing task.",
    "scoring": "Pearson correlation and Mean Maximum Rank Violation (MMRV) between DreamDojo and real-world success rates, following WorldEval and SIMPLER. Success is the share of 5 fruits picked from the table and placed in a bag, averaged over 20 scenes. Human evaluators score the generated rollouts with the same criteria as the real ones.",
    "licence": "Code: Apache-2.0 (github.com/NVIDIA/DreamDojo, GitHub API and README 'License' section); Data: unknown for the AgiBot fruit-packing data and real evaluation rollouts used here (looked at: paper, project page, repo README; the README lists released GR-1 post-training and evaluation datasets, licence not checked)",
    "url": "https://arxiv.org/abs/2602.06949",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949",
      "title": "DreamDojo full text (arXiv HTML v1), Sec. 4.7 'Downstream Applications' and Limitations",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949v1/correlation_compressed.svg",
      "title": "DreamDojo Fig. 5(a): real vs DreamDojo success rates (points A-F)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://dreamdojo-world.github.io/",
      "title": "DreamDojo project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/NVIDIA/DreamDojo",
      "title": "NVIDIA/DreamDojo repository (GitHub API licence + README)",
      "type": "repo",
      "date": "2026-02-09",
      "accessed": "2026-10-11"
     }
    ],
    "note": "DreamDojo is an action-conditioned video world model pretrained on 44k hours of egocentric human video and post-trained per robot; DreamDojo-2B is used here. The evaluated policy is a single-view, state-free variant of GR00T N1.5. One task only. The GitHub repo description says ICML 2026."
   },
   {
    "name": "WorldArena",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Tsinghua University (lead; corresponding author Yong Li), Shanghai Jiao Tong University, The University of Hong Kong, Princeton University, Chinese Academy of Sciences, University of Science and Technology of China, Peking University, National University of Singapore",
    "date": "2026-02-09",
    "measures": "How well embodied world models (text- or action-conditioned robot video models) predict bimanual manipulation video, and whether they are useful for three downstream jobs: generating training data for a policy (data engine), standing in for a simulator when scoring policies (policy evaluator), and planning actions (action planner). Human ratings are collected as a third view.",
    "scoring": "Video quality uses 16 metrics in six sub-dimensions (visual quality, motion quality, content consistency, physics adherence, 3D accuracy, controllability), computed with MUSIQ, the LAION aesthetic predictor, V-JEPA feature MMD, RAFT optical flow, DINO and CLIP features, SAM 3 arm tracking with normalised DTW, monocular depth, and Qwen3-VL-8B as a 1-5 Likert judge (normalised to 0-1) for interaction quality, perspectivity and instruction following. Each metric is normalised to [0,1] (some with empirical 1st/99th-percentile bounds) and scaled to 0-100; EWMScore is the arithmetic mean of the normalised metrics. Data engine: success rate of a pi0.5 policy trained on 25 generated trajectories per task (2 tasks, 100 trials each). Policy evaluator: Pearson correlation between VLM-judged success rates of 5 pi0.5 policies rolled out inside the world model and their RoboTwin simulator success rates. Action planner: success rate in RoboTwin when the world model plus an inverse dynamics model outputs actions. Human evaluation: 1-5 scores on overall quality, instruction following and physical adherence (normalised to 0-100) plus head-to-head win rate.",
    "licence": "Code: no licence file at the root of github.com/tsinghua-fib-lab/WorldArena (GitHub API license: null); subfolders carry their own: embodied_task/LICENSE is MIT (Copyright 2025 Tsinghua University) and embodied_task/worldarena_track2/LICENSE is Apache-2.0 (Copyright 2026 WorldArena Team); the leaderboard Space card (WorldArena/WorldArena) declares MIT. Data: Apache-2.0 on the HF dataset card WorldArena/WorldArena_Robotwin2.0 (a curated subset of RoboTwin 2.0). Website text: CC BY-SA 4.0 (world-arena.ai footer).",
    "url": "https://world-arena.ai",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2602.08971",
      "title": "WorldArena: A Unified Benchmark for Evaluating Perception and Functional Utility of Embodied World Models (arXiv abs, v1 2026-02-09, v2 2026-02-11)",
      "type": "paper",
      "date": "2026-02-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/EWMScore_14models.svg",
      "title": "WorldArena Figure 1a: EWMScore of 14 models",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-arena.ai",
      "title": "WorldArena project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/README.md",
      "title": "WorldArena GitHub README (tsinghua-fib-lab/WorldArena)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/assets/README_submission.md",
      "title": "WorldArena evaluation and submission guideline",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.github.com/repos/tsinghua-fib-lab/WorldArena",
      "title": "GitHub API record for tsinghua-fib-lab/WorldArena (license: null; 269 stars; created 2026-02-11)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/embodied_task/LICENSE",
      "title": "WorldArena embodied_task/LICENSE (MIT, Copyright 2025 Tsinghua University)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/embodied_task/worldarena_track2/LICENSE",
      "title": "WorldArena embodied_task/worldarena_track2/LICENSE (Apache-2.0, Copyright 2026 WorldArena Team)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/datasets/WorldArena/WorldArena_Robotwin2.0/resolve/main/README.md",
      "title": "HF dataset card WorldArena/WorldArena_Robotwin2.0 (license: apache-2.0)",
      "type": "dataset-card",
      "date": "2026-07-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/spaces/WorldArena/WorldArena",
      "title": "HF Space WorldArena/WorldArena metadata (official leaderboard; card license: mit; last modified 2026-07-15)",
      "type": "leaderboard",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena.hf.space",
      "title": "WorldArena leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/spaces/WorldArena/WorldArena/resolve/main/src/data_loader.py",
      "title": "WorldArena leaderboard source: src/data_loader.py (EWMScore computation)",
      "type": "leaderboard",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 14 world models in the paper (CogVideoX, Wan 2.2, Wan 2.6, Veo 3.1, Genie Envisioner, GigaWorld-0, TesserAct, Cosmos-Predict 2.5 text and action variants, WoW, RoboMaster, Vidar, IRASim, Ctrl-World). Data: RoboTwin 2.0 Clean-50, 50 tasks x 50 episodes (README: 40 train and 10 test episodes per task; paper: 2000 training and 500 test videos). Human study: 70 annotators rated 3,500 videos. Data engine and action planner tests cover 6 models on 2 tasks (adjust bottle, click bell); the policy evaluator test covers 2 action-conditioned models. Paper result: top EWMScore Wan 2.6 61.86, then Ctrl-World 59.70 and Veo 3.1 58.87 (Figure 1a). Leaderboard: yes, https://huggingface.co/spaces/WorldArena/WorldArena (embedded on world-arena.ai). Read on 2026-10-10 through the Space's Gradio API: Track 1 (video quality) loaded 127 entries, top UNIS 73.64 ('EWMScore_P'), SisyphusWorld 73.06, BWM-Fast 72.71; Track 2 Data Engine (8 entries) top DSCFuncWorld 66.00; Track 2 Policy Evaluator (19 entries) top WorldScape v0.2 99.53 (Pearson r x 100). Conflicts: the paper and site describe 16 metrics, but the leaderboard code (src/data_loader.py) lists 15 base metrics (Action Following is absent) and multiplies raw 0-1 values by 100 instead of using the paper's empirical bounds, so leaderboard EWMScore values are not directly comparable with the paper's Figure 1a (inferred from the code). The paper also says empirical bounds came from 'the 8 models' although 14 were evaluated. Venue: none shown on primary pages. GitHub stars: 269 (2026-10-10). The policy-evaluator judge in the released code defaults to Qwen/Qwen3-VL-32B-Instruct; the paper does not name the judge model."
   },
   {
    "name": "GWM-Robotics policy evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Runway",
    "date": "2026-02-27",
    "measures": "Whether progress scores of VLA policies rolled out inside Runway's GWM-Robotics world model rank the policies in the same order as their real-world RoboArena evaluations.",
    "scoring": "Each policy acts on frames generated by GWM-Robotics from initial conditions of earlier RoboArena evaluations. Human graders rate task progress on each simulated rollout (about 10 graders per rollout; over 16,000 ratings for 1,450 rollouts). Per-policy scores are compared with RoboArena real-world scores using Pearson correlation and MMRV, following PolaRiS.",
    "licence": "Code: none public; GWM-Robotics and its Python SDK are offered by access request (runway.com/research/introducing-runway-gwm-1); Data: none released; no paper or code is linked from the post or the GWM-1 announcement.",
    "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://runway.com/research/introducing-runway-gwm-1",
      "title": "Introducing Runway GWM-1 (GWM Worlds, GWM Avatars, GWM Robotics)",
      "type": "blog",
      "date": "2025-12-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Company blog post by Runway Robotics (Andy Chen, Lucas Eager Leavitt, Rik Heijdens, Robin Kahlow). GWM-1 was announced 2025-12-11 as an autoregressive model built on Gen-4.5; the post calls GWM-Robotics a variant of Gen-4.5. Robot: Franka Emika Panda, tasks from RoboArena. The post states the study tests ranking and does not test absolute success-rate calibration."
   },
   {
    "name": "GigaBrain Challenge 2026 - World Model Track (CVPR 2026 workshop)",
    "kind": "challenge",
    "domains": [
     "robotics"
    ],
    "org": "GigaAI (organizers Zheng Zhu, Xiaofeng Wang) with co-organizers from University of Hong Kong, Peking University, Shanghai Jiao Tong University, RoboChallenge/Dexmal and Horizon Robotics",
    "date": "2026-03",
    "measures": "World models as evaluators of a VLA policy (GigaBrain) on 8 real-robot manipulation tasks: video quality when replaying teleoperation actions, and whether closed-loop rollouts driven by policy actions reach the same outcome as the real-robot reference video.",
    "scoring": "(1) Generation quality: a metric suite comparing generated and ground-truth videos (references: WorldArena, PBench); actions come from teleoperation (typically 300-1000 steps). (2) World model as evaluator: three annotators per team per task, each assigned up to 10 videos, score each video 0-3 against the real-robot reference (3 = same action and final state, plausible physics; 2 = same final state but deformation or implausible physics; 1 = arm motion matches but final state differs, e.g. success vs failure; 0 = both differ). Per-task score = (3xN3 + 2xN2 + 1xN1)/N_total, with the ground-truth video count as denominator if fewer videos are submitted; final score = mean over 8 tasks. Three submission rounds; the best round counts. Scoring checks per-episode outcome agreement with the real reference video; it does not compute policy-level success-rate correlation or ranking agreement. How parts (1) and (2) combine into the leaderboard 'best_score' is not stated (looked at: track guide, rubric page, leaderboard page, baseline repo README).",
    "licence": "Code: Apache-2.0 (github.com/open-gigaai/CVPR-2026-Workshop-WM-Track); Data: custom 'GigaBrain Challenge 2026 Data & Model License Agreement' (gated Hugging Face dataset; non-commercial research, no redistribution)",
    "url": "https://gigaai-research.github.io/GigaBrain-Challenge-2026/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://gigaai-research.github.io/GigaBrain-Challenge-2026/",
      "title": "GigaBrain Challenge 2026 @ CVPR 2026 (tracks, schedule, final rankings, organizers)",
      "type": "project-page",
      "date": "2026",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://gigaai-research.github.io/GigaBrain-Challenge-2026/guide/world-model.html",
      "title": "GigaBrain Challenge 2026 World Model Track Guide",
      "type": "project-page",
      "date": "2026",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://gigaai-research.github.io/GigaBrain-Challenge-2026/guide/evaluation-rubric.html",
      "title": "World Model as Evaluator - Scoring Criteria (0-3 rubric and aggregation)",
      "type": "project-page",
      "date": "2026",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/spaces/open-gigaai/CVPR-2026-WorldModel-Track-LeaderBoard/raw/main/results.json",
      "title": "World Model Track leaderboard results.json (Hugging Face space open-gigaai/CVPR-2026-WorldModel-Track-LeaderBoard)",
      "type": "leaderboard",
      "date": "2026-07-01",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/open-gigaai/CVPR-2026-Workshop-WM-Track",
      "title": "open-gigaai/CVPR-2026-Workshop-WM-Track baseline and evaluation code (GitHub API licence: Apache-2.0; README)",
      "type": "repo",
      "date": "2026-03-06",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/open-gigaai/CVPR-2026-WorldModel-Track-Dataset",
      "title": "Hugging Face API: open-gigaai/CVPR-2026-WorldModel-Track-Dataset (gated; card holds 'GigaBrain Challenge 2026 Data & Model License Agreement')",
      "type": "dataset-card",
      "date": "2026-03-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Date is the competition kickoff (mid-March 2026); the competition ended May 15, 2026 and the workshop with awards was held June 3, 2026 (Colorado Convention Center, Denver). Other tracks: RoboTwin (simulation), RoboChallenge (real robot) and a PhysClaw demo track. Final World Model Track ranking: champion team xuwu (Zhenyu Wu, Xiuwei Xu, Ziwei Wang, Jiwen Lu; Tsinghua University; best score 57.10864806), 1st runner-up ABot-PhysWorld (AMAP CV Lab; 55.94211823), third Agent (Xiaomi EV; 54.35207991). Three champion-team members (Zhenyu Wu, Xiuwei Xu, Jiwen Lu) are also GigaWorld-1 co-authors. The leaderboard results.json lists 15 entries; GigaWorld-1 is the provided baseline."
   },
   {
    "name": "WorldArena Challenge @ CVPR 2026",
    "kind": "challenge",
    "domains": [
     "robotics"
    ],
    "org": "Organisers listed on the challenge page: Amap CV Lab (AMAP), Manifold AI, Tsinghua University, Princeton University, National University of Singapore, The University of Hong Kong",
    "date": "2026-03",
    "measures": "Two tracks built on WorldArena: Track 1 video perception quality of embodied world models; Track 2 whether generated worlds work as data engines and as policy evaluators.",
    "scoring": "Track 1: arithmetic mean of the WorldArena video metrics x 100 (EWMScore) on the official test set. Track 2 Data Engine: success rates of policies trained on generated data, averaged over task columns (result files list adjust_bottle, blocks_ranking_rgb, click_bell, open_laptop, pick_dual_bottles). Track 2 Policy Evaluator: participants roll out 5 fixed pi0.5 checkpoints for 500 episodes each inside their world model; a VLM judge (default Qwen3-VL-32B-Instruct) labels each rollout 0/1 against ground truth; the score is the Pearson r between the 5 success rates and the simulator success rates (28.60, 34.58, 37.78, 43.52, 46.80). At most two submissions per day; closed-source models had to provide API access.",
    "licence": "Same as WorldArena. Code: no root licence (subfolders MIT and Apache-2.0); Data: Apache-2.0 (HF dataset card WorldArena/WorldArena_Robotwin2.0).",
    "url": "http://cvpr2026challenge.world-arena.ai/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "http://cvpr2026challenge.world-arena.ai/",
      "title": "WorldArena Challenge @ CVPR 2026 page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://videoworldmodel-workshop.github.io/",
      "title": "1st Workshop on Video World Models (CVPR 2026) page, WorldArena Challenge section",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/README.md",
      "title": "WorldArena GitHub README (tsinghua-fib-lab/WorldArena)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/assets/README_submission.md",
      "title": "WorldArena evaluation and submission guideline",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/embodied_task/policy_eval_release_bundle/Policy_eval.md",
      "title": "WorldArena Track 2 policy evaluation with a VLM judge (Policy_eval.md)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/spaces/WorldArena/WorldArena",
      "title": "HF Space WorldArena/WorldArena metadata (official leaderboard; card license: mit; last modified 2026-07-15)",
      "type": "leaderboard",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena.hf.space",
      "title": "WorldArena leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Held with the CVPR 2026 Workshop on Video World Models (Denver). Prizes: Track 1 $3000 / $2000 / $1000; Track 2 $4000 / $2500 / $1500. Date conflicts across primary pages: the challenge page lists submission opening 2026-03-06, final deadline 2026-05-25 and the challenge session on 2026-06-04, but its header says June 25, 2026; the workshop page gives June 3, 2026 for the workshop; the GitHub README dates the challenge opening to 2026-03-26; the submission guide says the challenge concluded with a June 30, 2026 deadline; the leaderboard banner announces June 2026 update dates of 6.15 and 6.25. Winners were not listed on the challenge, workshop or repository pages checked. Leaderboard: https://huggingface.co/spaces/WorldArena/WorldArena; top on 2026-10-10: Track 1 UNIS 73.64; Track 2 Data Engine DSCFuncWorld 66.00; Track 2 Policy Evaluator WorldScape v0.2 99.53. Ground truth and evaluation resources are now public per the submission guide."
   },
   {
    "name": "PlayWorld policy evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Princeton University",
    "date": "2026-03-09",
    "measures": "Whether success rates predicted by a video world model trained on autonomous robot play match real-robot success rates for many manipulation policies on contact-rich tasks, and whether predicted failure modes match observed ones.",
    "scoring": "Pearson correlation and RMSE between world-model and real success rates across 18 policies; 20 real-world trials and 50 world-model trials per policy. Human-annotated failure-mode distributions are compared per policy.",
    "licence": "Code: no licence file found (github.com/irom-princeton/open-world, linked as 'Code' from the project page; GitHub API reports no licence); Data: no licence stated (Hugging Face dataset tennyyyin/playworld_dataset_preview, gated, card has no licence field)",
    "url": "https://arxiv.org/abs/2603.09030",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2603.09030",
      "title": "PlayWorld: Learning Robot World Models from Autonomous Play (arXiv abstract; v1 2026-03-09, v3 2026-04-06)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030",
      "title": "PlayWorld full text (arXiv HTML v3), Sec. 4.4 and Fig. 7",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v3/images/experiment/correlation_new.png",
      "title": "PlayWorld Fig. 7: policy evaluation success-rate correlation (RMSE and r per training-data source)",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v1",
      "title": "PlayWorld arXiv HTML v1 (checked that Pearson 0.8766 and 18 policies already appear)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://robot-playworld.github.io/",
      "title": "PlayWorld project page (links Code and Data)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/irom-princeton/open-world",
      "title": "irom-princeton/open-world repository (linked as Code; GitHub API shows no licence)",
      "type": "repo",
      "date": "2026-02-10",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/tennyyyin/playworld_dataset_preview",
      "title": "Hugging Face API: tennyyyin/playworld_dataset_preview (no licence field, gated)",
      "type": "dataset-card",
      "date": "2026-03-28",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Action-conditioned video model (pre-trained Stable Video Diffusion backbone, initialised from a DROID-pretrained checkpoint) fine-tuned on 30 hours of task-agnostic autonomous play data on a DROID manipulation setup. Baselines use the same architecture trained on human demonstration data or human play data. Three object sets / tasks. Name collision: not the same as PlayWorld (arXiv 2608.13552), a 2026 benchmark for interactive game world models."
   },
   {
    "name": "Interactive World Simulator sim-to-real policy evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Columbia University; Toyota Research Institute; Amazon; University of Illinois Urbana-Champaign",
    "date": "2026-03-09",
    "measures": "Whether task scores of imitation policies (final and intermediate checkpoints of DP, ACT, pi0 and pi0.5) evaluated closed-loop inside the learned world model match their real-robot scores on four ALOHA tasks.",
    "scoring": "Per-task correlation coefficient r between world-simulator and real task scores, each averaged over 20 initial configurations; Clopper-Pearson confidence intervals as error bars. Task scores: T pushing = maximum IoU with the target pose within 600 steps; rope routing = clips threaded within 200 steps; mug grasping = 1 point for grasp and 1 for placing within 200 steps; pile sweeping = pieces swept into the tray within 200 steps.",
    "licence": "Code: MIT licence text with an added attribution clause from Boyuan Chen's research template (LICENSE in github.com/WangYixuan12/interactive_world_sim; GitHub API reports 'Other'); Data: unknown (looked at: project page, repo root listing)",
    "url": "https://arxiv.org/abs/2603.08546",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2603.08546",
      "title": "Interactive World Simulator for Robot Policy Training and Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.08546",
      "title": "Interactive World Simulator full text (arXiv HTML v1), Sec. IV-A, IV-C, IV-D",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.08546v1/correlation_v3.png",
      "title": "Interactive World Simulator Fig. 7: world-simulator vs real task scores per task (r values)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://roboticsproceedings.org/rss22/p018.html",
      "title": "RSS XXII (2026) proceedings page p018, DOI 10.15607/RSS.2026.XXII.018",
      "type": "paper",
      "date": "2026-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://roboticsproceedings.org/rss22/p018.pdf",
      "title": "RSS 2026 paper PDF p018 (Fig. 7 r values)",
      "type": "paper",
      "date": "2026-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.yixuanwang.me/interactive_world_sim/",
      "title": "Interactive World Simulator project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/WangYixuan12/interactive_world_sim",
      "title": "WangYixuan12/interactive_world_sim repository (LICENSE file; GitHub API reports 'Other')",
      "type": "repo",
      "date": "2026-02-10",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Published in Robotics: Science and Systems XXII (RSS 2026, Sydney, July 2026), DOI 10.15607/RSS.2026.XXII.018. Robot: ALOHA bimanual robot. The world model uses consistency models for image decoding and latent dynamics and is trained per task on about 600 play episodes of 200 steps."
   },
   {
    "name": "PersistWorld world-model-to-real policy evaluation",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Czech Institute of Informatics, Robotics and Cybernetics, Czech Technical University in Prague",
    "date": "2026-03-26",
    "measures": "Whether the task progress of robot policies rolled out inside an action-conditioned video world model matches their real-robot task progress, comparing PersistWorld (Ctrl-World post-trained with reinforcement learning on its own rollouts) with the base Ctrl-World.",
    "scoring": "Pearson r and MMRV over 9 task-policy pairs. Task progress uses a step rubric per task (for example Reach, Grasp, Lift, Move Close, IsInside), each completed step worth 1/N, averaged over 5 real and 11 world-model rollouts per task-policy pair.",
    "licence": "Code: MIT (github.com/Jai2500/PersistWorld, GitHub API); Data: unknown (looked at: project page, repo metadata)",
    "url": "https://arxiv.org/abs/2603.25685",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2603.25685",
      "title": "Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning (arXiv abstract; v1 2026-03-26, v2 2026-09-04; ECCV 2026)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v1",
      "title": "PersistWorld arXiv HTML v1, Appendix 0.C (same Fig. 8 file as v2)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Jai2500/PersistWorld",
      "title": "Jai2500/PersistWorld repository (GitHub API licence: MIT)",
      "type": "repo",
      "date": "2026-03-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://jaibardhan.com/persistworld/",
      "title": "PersistWorld project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Accepted at ECCV 2026 (arXiv comments). The policy-evaluation appendix and Fig. 8 are already in v1 (2026-03-26); v2 (2026-09-04) adds the numbers to the main text. The world model is evaluated on DROID (Franka Emika Panda)."
   },
   {
    "name": "Cosmos-H-Surgical-Simulator evaluation on Open-H-Embodiment",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "NVIDIA with the Open-H-Embodiment Consortium (50+ institutions in the paper's author list); model card by NVIDIA",
    "date": "2026-04",
    "measures": "How closely an action-conditioned surgical world model's generated video follows the recorded kinematic actions of held-out real surgical-robot episodes (open-loop replay), as groundwork for in-silico policy evaluation.",
    "scoring": "Model card: FDS (L1), mean L1 distance between generated and ground-truth frames normalised to [-1, 1], lower is better; GATC, median zero-mean normalised cross-correlation of grayscale pixels inside SAM3-segmented tool regions, weighted by a tool-presence penalty, higher is better; TCD, median per-frame Euclidean distance in pixels between Hungarian-matched tool centroids with a half-diagonal penalty for unmatched tools, lower is better. Card setup: 4 CMR Versius procedures at 360p, 2 episodes per procedure, 2 seeds, 72-frame autoregressive generation (6 chunks x 12 frames). Paper setup: per-frame L1 and SSIM over 72 frames on held-out episodes from 25 of 32 training datasets, 2 episodes per dataset, 3 seeds (up to 150 episode evaluations), split into benchtop (18 datasets) and tissue-based (7 datasets).",
    "licence": "Code: Apache-2.0 (github.com/NVIDIA-Medtech/Cosmos-H-Surgical-Simulator). Model weights: NVIDIA Open Model License (HF model card). Data: CC-BY-4.0 (HF dataset nvidia/PhysicalAI-Robotics-Open-H-Embodiment; also stated in the paper).",
    "url": "https://huggingface.co/nvidia/Cosmos-H-Surgical-Simulator",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://huggingface.co/nvidia/Cosmos-H-Surgical-Simulator/resolve/main/README.md",
      "title": "HF model card nvidia/Cosmos-H-Surgical-Simulator (Quantitative Evaluation: FDS, GATC, TCD)",
      "type": "dataset-card",
      "date": "2026-10-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2604.21017",
      "title": "Open-H-Embodiment: A Large-Scale Dataset for Enabling Foundation Models in Medical Robotics (arXiv abs; v1 2026-04-22, v3 2026-06-04)",
      "type": "paper",
      "date": "2026-04-22",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.21017v3",
      "title": "Open-H-Embodiment full text (arXiv HTML v3), Cosmos-H-Surgical-Simulator sections",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://api.github.com/repos/NVIDIA-Medtech/Cosmos-H-Surgical-Simulator",
      "title": "GitHub API record for NVIDIA-Medtech/Cosmos-H-Surgical-Simulator (license Apache-2.0)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/nvidia/PhysicalAI-Robotics-Open-H-Embodiment",
      "title": "HF dataset nvidia/PhysicalAI-Robotics-Open-H-Embodiment metadata (license: cc-by-4.0)",
      "type": "dataset-card",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Date: the card's April 2026 update reports the three metrics; paper v1 2026-04-22 (v3 2026-06-04). Card results, current checkpoint: FDS 0.184, GATC 0.472, TCD 67.03 px (previous checkpoint 0.223, 0.417, 83.68); per procedure TCD ranges from 12.7 px (hysterectomy) to 143.2 px (inguinal hernia). Conflict: the paper (v3) says segmentation-based metrics such as tool consistency and tool centroid distance 'proved insufficiently robust' across embodiments and reports only L1 and SSIM, while the card (last modified 2026-10-09) reports GATC and TCD. No agreement with real policy outcomes is reported for this model; the paper cites prior single-platform work (Cosmos-Surg-dVRK) for that. Leaderboard: none."
   },
   {
    "name": "RoboWM-Bench",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Peking University (lead; corresponding author Ruihai Wu), Tsinghua University, Lightwheel",
    "date": "2026-04-21",
    "measures": "Whether manipulation behaviour in videos generated by world models can be executed: generated human-hand or robot-arm videos are converted to robot actions and run in simulated scenes, including real-to-sim reconstructions of real tabletop scenes.",
    "scoring": "Human-hand videos: 3D hand-pose tracking and retargeting to robot end-effector actions. Robot videos: an inverse dynamics model predicts joint-space action chunks. Actions run on a Franka arm in simulation (LeHome engine on Isaac Lab); each task uses 10 initial object configurations shared across models; outputs are task-level success rates and step-level checker rates (e.g., contact, lift, place).",
    "licence": "Code: none found (no LICENSE at the root of github.com/fffstrong/RoboWM-Bench; GitHub API license: null; the vendored IsaacLab_5_1 folder keeps Isaac Lab's own licence files); IDM weights at HF RoboWM/RoboWM-IDM-real have no licence in the card. Data: no separate dataset licence found (task inputs ship in the repository). Project page text: CC BY-SA 4.0.",
    "url": "https://robowm-bench.github.io/RoboWM-Bench/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.19092",
      "title": "RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation (arXiv abs; v1 2026-04-21, v2 2026-05-14)",
      "type": "paper",
      "date": "2026-04-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.19092v2",
      "title": "RoboWM-Bench full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://robowm-bench.github.io/RoboWM-Bench/",
      "title": "RoboWM-Bench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/fffstrong/RoboWM-Bench/HEAD/README.md",
      "title": "fffstrong/RoboWM-Bench README (links arXiv 2604.19092 and the project page)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://api.github.com/repos/fffstrong/RoboWM-Bench",
      "title": "GitHub API record for fffstrong/RoboWM-Bench (license: null; created 2026-04-15)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Repository github.com/fffstrong/RoboWM-Bench is linked from the project page and its README links the arXiv paper. Scale (Table 1): 8 human-hand tasks and 8 real-to-sim robot tasks (simulation-native tasks in the appendix). Models: Cosmos, Wan 2.2, Veo 3.1, Wan 2.6 and LVP on human-hand tasks; Cosmos, Wan 2.2, Veo 3.1, Wan 2.6 and Cosmos fine-tuned on 50 trajectories per task (Cosmos-FT) on robot tasks. Results: Wan 2.6 is the strongest overall (human-hand task success 40% to 100%); robot-video success is lower (Wan 2.6 0% to 50%; Cosmos-FT 20% to 90%). Venue: Best Paper Award, CVPR 2026 Workshop on World Models Meet Active Sensing and Closed-Loop Planning (WMAS), per the project page. Leaderboard: none (results table only). GitHub stars: 42."
   },
   {
    "name": "dWorldEval policy evaluation proxy",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Current Robotics; University of Toronto",
    "date": "2026-04-24",
    "measures": "Whether success rates estimated inside a discrete-diffusion world model (with a progress token that marks task completion) match ground-truth success for pi0 checkpoints on LIBERO, heterogeneous policies on RoboTwin, and three policies on five real bimanual AgileX tasks; it also compares three video-diffusion world models on LIBERO.",
    "scoring": "Pearson r and MMRV between world-model and ground-truth success rates. Ground truth is the simulator outcome for LIBERO and RoboTwin and real execution for AgileX. In the world model a rollout succeeds when the predicted progress token reaches 1 (the no-memory ablation is judged on the generated image). 20 episodes per task in simulation and 30 in the real world.",
    "licence": "Code: unknown (no repository link on the project page or arXiv page); Data: unknown (looked at: arXiv abstract, full text, project page)",
    "url": "https://arxiv.org/abs/2604.22152",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.22152",
      "title": "dWorldEval: Scalable Robotic Policy Evaluation via Discrete Diffusion World Model (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2604.22152v1",
      "title": "dWorldEval arXiv PDF v1, first page (author-to-affiliation mapping)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://dworldeval.github.io/",
      "title": "dWorldEval project page (affiliations; 'Accepted by ICML 2026 Spotlight')",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Affiliations from the arXiv PDF and project page: Yaxuan Li, Zhongyi Zhou, Yefei Chen and Yaokai Xue at Current Robotics; Yichen Zhu at University of Toronto. Project page says 'Accepted by ICML 2026 Spotlight'; its BibTeX block shows the WorldEval entry. Model initialised from MMaDA-VLA-8B. Baselines (WorldEval, WorldGym, Ctrl-World) were trained by the dWorldEval authors on identical data splits."
   },
   {
    "name": "Cosmos-HumanEval (Cosmos HUE)",
    "kind": "protocol",
    "domains": [
     "general-video",
     "robotics"
    ],
    "org": "NVIDIA",
    "date": "2026-05",
    "measures": "Human judgement of physical laws, visual integrity, semantic alignment and geometry in generated videos.",
    "scoring": "Two annotators answer up to 20 single-fact Yes / No / Unclear questions per video (Yes is always the desirable outcome), grouped into four dimensions; results are pass rates, with real videos scored as an upper reference.",
    "licence": "Data: custom OpenMDW-1.1 with a no-AI-training clause (HF card); Code: unknown",
    "url": "https://huggingface.co/datasets/nvidia/Cosmos-HumanEval-v1",
    "in_atlas": true,
    "atlas_id": "cosmos-humaneval",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "report",
      "date": "2026-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/datasets/nvidia/Cosmos-HumanEval-v1",
      "title": "nvidia/Cosmos-HumanEval-v1 dataset card",
      "type": "dataset-card",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'cosmos-humaneval' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "WorldArena 2.0",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Tsinghua University (lead; corresponding author Yong Li), Shanghai Jiao Tong University, Zhejiang University, Stanford University, The University of Hong Kong, Princeton University, Chinese Academy of Sciences, University of Science and Technology of China, Peking University, National University of Singapore",
    "date": "2026-05-18",
    "measures": "Extends WorldArena along three axes: visuotactile prediction (tactile plus video) for contact-rich tasks, world models used as interactive reinforcement-learning environments for policy improvement, and evaluation across two simulators and a real robot (RoboTwin 2.0, LIBERO, AgileX split-type ALOHA).",
    "scoring": "Visuotactile: PSNR and SSIM of predicted tactile signals plus task success on Insert HDMI and Lift Bottle in the UniVTAC simulator, using a standard pipeline that adds a tactile VAE, a two-stream visuotactile model and an action diffusion head to existing video world models. RL environment: success rate of a pi0.5 policy after RL inside each world model on Click Bell and Adjust Bottle, with three reward models (ResNet proxy, Qwen-3.5 VLM, visual similarity), compared with SFT and simulator-based RL. Cross-platform: WorldArena's 16 video metrics on each platform, and data-engine and action-planner success rates on RoboTwin, LIBERO and two real tasks (pour water, wipe table).",
    "licence": "Code: MIT (LICENSE at the root of github.com/WorldArena2/WorldArena-2.0, per GitHub API). Data: Apache-2.0 (HF dataset WorldArena/WorldArena2.0 card). Leaderboard Space card: MIT. Website text: CC BY-SA 4.0 (worldarena2.github.io footer).",
    "url": "https://worldarena2.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2605.17912",
      "title": "WorldArena 2.0: Extending Embodied World Model Benchmarking on Modality, Functionality and Platform (arXiv abs)",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2605.17912",
      "title": "WorldArena 2.0 full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena2.github.io/",
      "title": "WorldArena 2.0 project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.github.com/repos/WorldArena2/WorldArena-2.0",
      "title": "GitHub API record for WorldArena2/WorldArena-2.0 (license MIT; created 2026-05-05)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/WorldArena/WorldArena2.0",
      "title": "HF dataset WorldArena/WorldArena2.0 metadata (card license: apache-2.0)",
      "type": "dataset-card",
      "date": "2026-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/spaces/WorldArena/WorldArena2.0",
      "title": "HF Space WorldArena/WorldArena2.0 metadata (card license: mit; last modified 2026-09-18)",
      "type": "leaderboard",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena2-0.hf.space",
      "title": "WorldArena 2.0 leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: the paper reports experiments on 12 embodied world models; visuotactile test covers Vidar, Wan2.2, Genie Envisioner and an ACT baseline (Wan2.2 100% on Insert HDMI; all world models 0% on Lift Bottle versus ACT 80%); RL test covers 7 world models (WoVR best on Click Bell, 75.00 with proxy reward; Ctrl-World best on Adjust Bottle, 70.70; simulator-based RL 87.30 and 78.90; SFT 43.75 and 55.08); cross-platform success table covers 6 models. Leaderboard: yes, https://huggingface.co/spaces/WorldArena/WorldArena2.0, read 2026-10-10 through its Gradio API: Track 1 (simulator video quality, 'EWMScore-P (Difficulty/OOD Adjusted)', 81 entries) top WorldIncept 72.33; Track 2 (world model as RL environment, Adjust Bottle success, 47 entries) top MW2 72.93; Track 3 (real robot, overall, 11 entries) top ViTacX 76.64. The leaderboard serves the WorldArena 2.0 Challenge at IROS 2026 (separate row). Venue: none shown."
   },
   {
    "name": "OSCAR policy evaluation on RoboArena",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Peking University; University of Michigan; NVIDIA",
    "date": "2026-06-03",
    "measures": "Whether OSCAR-generated replays of RoboArena episodes, judged by GPT-5, reproduce the real per-policy success rates and ranking of seven open-source DROID policies.",
    "scoring": "MMRV on the seven-policy rank vector (range [0,6]), Spearman rho between rank vectors, Pearson r between per-policy mean binary success rates, and SISR_delta (mean absolute error in percentage points); 65 sessions x 7 policies; GPT-5 judge on 32 sampled frames; 1,365 GPT-5 pairwise preferences turned into Bradley-Terry scores.",
    "licence": "unknown (looked at: arXiv abstract and full text; project page not opened)",
    "url": "https://arxiv.org/abs/2606.04463",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.04463",
      "title": "OSCAR: Omni-Embodiment Action-Conditioned World Model for Robotics (arXiv abstract; v1 2026-06-03, v2 2026-06-04)",
      "type": "paper",
      "date": "2026-06-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2606.04463",
      "title": "OSCAR full text (arXiv HTML v2), Sec. 5.4 Table 4, Appendix A.10",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": "OSCAR fine-tunes Cosmos-Predict2.5-2B with 2D kinematic skeleton renderings as the action condition. Each episode is rolled out from the recorded first frame with the recorded robot actions. Real outcomes come from RoboArena (DROID platform: Franka Panda with Robotiq gripper). MMRV here uses ranks, so its scale differs from the [0,1] MMRV in other papers."
   },
   {
    "name": "WEAVER policy evaluation on real hardware",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Mila - Quebec AI Institute; Universite de Montreal; Carnegie Mellon University; McGill University",
    "date": "2026-06-11",
    "measures": "Whether success judged on rollouts imagined by the world model matches the real success rates of two pi0.5-based policies on five real manipulation tasks, compared with Ctrl-World and with the WEAVER model before task fine-tuning.",
    "scoring": "Pearson, Spearman, RMSE and MMRV between world-model and real success rates over 10 task-policy points. Real success comes from 20 held-out trials per task. Imagined rollouts replay recorded real action sequences open-loop (Sec. 3.4) and humans label binary success.",
    "licence": "Code: MIT (github.com/arnavkj1995/WEAVER, GitHub API); Model weights: no licence field on Hugging Face arnavkj1995/WEAVER; Data: no licence field on Hugging Face yilin-wu/droid_ood_data",
    "url": "https://arxiv.org/abs/2606.13672",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.13672",
      "title": "WEAVER, Better, Faster, Longer: An Effective World Model for Robotic Manipulation (arXiv abstract; v1 2026-06-11, v2 2026-06-16)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v2/policy_eval.png",
      "title": "WEAVER Fig. 6: policy evaluation scatter plots (rho and MMRV per world model)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v1",
      "title": "WEAVER arXiv HTML v1 (checked that rho=0.870 and Table 8 values already appear)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arnavkj1995.github.io/WEAVER/",
      "title": "WEAVER project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/arnavkj1995/WEAVER",
      "title": "arnavkj1995/WEAVER repository (GitHub API licence: MIT)",
      "type": "repo",
      "date": "2026-05-08",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/models/arnavkj1995/WEAVER",
      "title": "Hugging Face API: arnavkj1995/WEAVER model (no licence field)",
      "type": "dataset-card",
      "date": "2026-06-08",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/yilin-wu/droid_ood_data",
      "title": "Hugging Face API: yilin-wu/droid_ood_data (no licence field)",
      "type": "dataset-card",
      "date": "2026-06-10",
      "accessed": "2026-10-11"
     }
    ],
    "note": "WEAVER is a 928M-parameter multi-view latent world model with reward and critic heads, pre-trained on DROID and fine-tuned on 50 pi0.5 rollouts per task (WEAVER-FT). Hardware: DROID setup with one Franka Emika Panda. Rollouts last up to 40 seconds."
   },
   {
    "name": "RoboWorld",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "KAIST; Config",
    "date": "2026-07-01",
    "measures": "Whether scores from closed-loop rollouts of DROID policies in an autoregressive video world model match the RoboArena real-world leaderboard.",
    "scoring": "World model adapted from Wan2.1-T2V-1.3B with Step Forcing; 30-second rollouts from RoboArena initial frames; GPT-4o scores each rollout on a 0-5 task-progress rubric that separates world-model errors from policy failures; Pearson r and Spearman rho against the leaderboard.",
    "licence": "Code: not released (project page says 'Code coming soon'); Data: unknown (looked at the project page and paper).",
    "url": "https://byeongguks.github.io/RoboWorld/",
    "in_atlas": true,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2607.01060",
      "title": "RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation (arXiv abs; v1 2026-07-01, v4 2026-07-15)",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4)",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://byeongguks.github.io/RoboWorld/",
      "title": "RoboWorld project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Atlas entry roboworld. The project page lists the ICML 2026 F2S Workshop on Long-Horizon Video Generation."
   },
   {
    "name": "WMBench (GigaWorld-1)",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "GigaAI; Tsinghua University",
    "date": "2026-07-02",
    "measures": "How well video world models serve as surrogate evaluators of robot policies, by comparing generated rollouts with matched real-robot executions on eight manipulation tasks built from teleoperation data and GigaBrain policy rollouts.",
    "scoring": "Outcome score WMES per rollout on a 0-3 scale (3 = correct outcome and high fidelity; 2 = correct outcome, low fidelity; 1 = wrong outcome, high fidelity; 0 = wrong outcome and collapse), scored by three human annotators with senior spot checks or by a LoRA-tuned Qwen3-VL-8B judge. Automatic diagnostics from WorldArena (Aesthetic Quality, Image Quality, JEPA Similarity, Semantic Alignment, Subject Consistency, Trajectory Accuracy) are averaged into a summary score; long-horizon PSNR, FID and FVD are reported per 8-second interval; closed-loop success rates are compared with real robots at task and subtask level.",
    "licence": "Code: Apache-2.0 (github.com/open-gigaai/giga-world-1 LICENSE; challenge baseline repo open-gigaai/CVPR-2026-Workshop-WM-Track also Apache-2.0); Model: apache-2.0 (Hugging Face open-gigaai/Giga-World-1); Data: custom 'GigaBrain Challenge 2026 Data & Model License Agreement' on the gated Hugging Face dataset open-gigaai/CVPR-2026-WorldModel-Track-Dataset (academic and non-commercial research use, no redistribution; licence field empty)",
    "url": "https://open-gigaai.github.io/giga-world-1/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2607.02642",
      "title": "GigaWorld-1: A Roadmap to Build World Models for Robot Policy Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://open-gigaai.github.io/giga-world-1/",
      "title": "GigaWorld-1 project page (WMBench leaderboard table)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/open-gigaai/giga-world-1",
      "title": "open-gigaai/giga-world-1 repository (LICENSE: Apache-2.0)",
      "type": "repo",
      "date": "2026-06-30",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/open-gigaai/CVPR-2026-WorldModel-Track-Dataset",
      "title": "Hugging Face API: open-gigaai/CVPR-2026-WorldModel-Track-Dataset (gated; card holds 'GigaBrain Challenge 2026 Data & Model License Agreement')",
      "type": "dataset-card",
      "date": "2026-03-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/models/open-gigaai/Giga-World-1",
      "title": "Hugging Face API: open-gigaai/Giga-World-1 model (licence apache-2.0)",
      "type": "dataset-card",
      "date": "2026-07-01",
      "accessed": "2026-10-11"
     }
    ],
    "note": "2,989 paired trajectories across eight tasks; teleoperation and GigaBrain rollout data about 1:1; training set 82,470 s, test set 7,200 s. Serves as the official benchmark of the CVPR 2026 GigaBrain Challenge World Model Track. The paper says the Hugging Face dataset had over 50,000 downloads; the Hugging Face API showed downloads = 70 on 2026-10-11 (a recent-period count). The abstract reports 7 video world models, 4 action representation schemes and 'over 324,000 simulated policy rollouts paired with real robot executions'; Sec. 4.1 describes these as 324,000 world-model rollout segments sampled from submissions of 'over 100 participating teams' and chained into episodes of about 20-30 segments. Project-page leaderboard (AVG of six metrics): GigaWorld-1-Plus 0.6834, GigaWorld-1-Nano 0.6716, Cosmos-Predict2.5 0.6123, Wan 2.2 5B 0.5948, LTX 2.3 0.5775, CogVideoX 0.5620, SVD 0.5569, Wan 2.1 1.3B I2V 0.5355; the paper text gives Nano 0.6717. The project-page figure caption mentions seven metrics and calls Nano best overall, which conflicts with its own table and the paper."
   },
   {
    "name": "WorldArena 2.0 Challenge @ IROS 2026",
    "kind": "challenge",
    "domains": [
     "robotics"
    ],
    "org": "WorldArena 2.0 team (organisers not named on the challenge page; contact worldarenav2@outlook.com); paper team led by Tsinghua University",
    "date": "2026-07-10",
    "measures": "Three tracks: Track 1 video quality of embodied world models with harder tasks and new out-of-distribution scenes; Track 2 world models as RL environments for policy optimisation; Track 3 real-world manipulation by world action models (WAM), in tactile and vision-only settings.",
    "scoring": "Track 1: difficulty- and OOD-adjusted EWMScore ('EWMScore-P') from the WorldArena perceptual metrics. Track 2: success rate of policies trained inside each world model (leaderboard column: Adjust Bottle). Track 3: real-robot success rates per task (wipe table, pour water, clean tabletop, instruction-following clean tabletop, fold clothes, fold cardboard box; visuo-tactile pick potato chip, peel cucumber, insert two-pin plug), averaged into a track score. At most two submissions per team per day.",
    "licence": "Code: MIT (github.com/WorldArena2/WorldArena-2.0). Data: Apache-2.0 (HF dataset WorldArena/WorldArena2.0).",
    "url": "http://iros2026challenge.world-arena.ai/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "http://iros2026challenge.world-arena.ai/",
      "title": "WorldArena 2.0 Challenge, IROS 2026 Competition page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena2-0.hf.space",
      "title": "WorldArena 2.0 leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/spaces/WorldArena/WorldArena2.0",
      "title": "HF Space WorldArena/WorldArena2.0 metadata (card license: mit; last modified 2026-09-18)",
      "type": "leaderboard",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://api.github.com/repos/WorldArena2/WorldArena-2.0",
      "title": "GitHub API record for WorldArena2/WorldArena-2.0 (license MIT; created 2026-05-05)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/WorldArena/WorldArena2.0",
      "title": "HF dataset WorldArena/WorldArena2.0 metadata (card license: apache-2.0)",
      "type": "dataset-card",
      "date": "2026-07-29",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Dates on the page: competition opens July 10, 2026; leaderboard updates July 30 and August 15; final submission deadline August 30; final results September 15; award ceremony September 27, 2026 at an IROS 2026 workshop. Prizes: Track 1 $700 / $400 / $300; Tracks 2 and 3 $1,400 / $800 / $600 each. Leaderboard (read 2026-10-10): Track 1 WorldIncept 72.33 (81 entries); Track 2 MW2 72.93 (47 entries); Track 3 overall ViTacX 76.64 (11 entries); Track 3.1 visuo-tactile Nexus-T 80.00; Track 3.2 vision-only ViTacX 85.00. The Track 3 tab states it shows results completed and verified by the August 15 checkpoint. Final winners were not found on the challenge page."
   },
   {
    "name": "TriWorldBench",
    "kind": "benchmark",
    "domains": [
     "robotics"
    ],
    "org": "Peking University, Tsinghua University, Beihang University, Shanghai Jiao Tong University, University of Science and Technology of China, Shanghai AI Laboratory (arXiv affiliation block); the website lists OpenCompass in place of Shanghai AI Laboratory",
    "date": "2026-07-28",
    "measures": "Whether an embodied world model's synchronized head, left-wrist and right-wrist videos describe one consistent manipulation event, plus task alignment, physical and 3D coherence, motion quality, temporal consistency and visual quality.",
    "scoring": "19 metrics in six dimensions: tri-view consistency (normalised PSNR and SSIM per camera against ground truth, three VLM consistency rubrics on head-wrist pairs, VQA accuracy on a fixed question bank), task alignment (instruction following, semantic alignment of captions, V-JEPA similarity), physical and 3D coherence (interaction quality, perspective), motion quality (state alignment, flow score, trajectory accuracy), temporal consistency and visual quality (with penalty-adjusted variants). VLM metrics use Qwen3-VL-8B-Instruct. Reference action phases derived from trajectories set expected motion per view. Metrics are averaged within each dimension and the six dimension scores combine into TWB-Score (0-100).",
    "licence": "Code: none (no LICENSE at the root of github.com/TriWorldBench/TriWorldBench; GitHub API license: null). Data: none stated (HF dataset card TriWorldBench/Dataset has no license field); episodes are derived from RoboTwin 2.0.",
    "url": "https://www.triworldbench.com/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2609.26314",
      "title": "TriWorldBench: A Tri-View Consistency Perspective on Embodied World Models (arXiv abs)",
      "type": "paper",
      "date": "2026-09-22",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2609.26314",
      "title": "TriWorldBench full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-09-22",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/TriWorldBench/TriWorldBench/HEAD/README.md",
      "title": "TriWorldBench GitHub README (news: code, val and test sets released and challenge launched 2026-07-28)",
      "type": "repo",
      "date": "2026-07-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/TriWorldBench/TriWorldBench/HEAD/DOWNLOAD_LINKS.md",
      "title": "TriWorldBench metric weights guide (Qwen3-VL-8B-Instruct judge)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://api.github.com/repos/TriWorldBench/TriWorldBench",
      "title": "GitHub API record for TriWorldBench/TriWorldBench (license: null; 235 stars)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/TriWorldBench/Dataset/resolve/main/README.md",
      "title": "HF dataset card TriWorldBench/Dataset (no license field)",
      "type": "dataset-card",
      "date": "2026-07-28",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.triworldbench.com/",
      "title": "TriWorldBench website with leaderboard (55 published models)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Date: code, 100-episode validation set and 500-episode test set released and the TriWorldBench Challenge launched on 2026-07-28 (README news); arXiv v1 is 2026-09-22. Scale: 500 test episodes over the 50 bimanual RoboTwin 2.0 tasks, with clean and domain-randomised backgrounds; separate 100-episode validation split derived from RoboTwin 2.0. The paper reports no model results. Leaderboard: yes, on www.triworldbench.com, 55 published models (accessed 2026-10-11); top BWM-Pro TWB-Score 70.42, second 'digger' 70.23. Submissions are evaluated in two-week cycles (current cycle ends 2026-10-11, results 2026-10-16, Beijing time); the rules exclude videos with text added to influence the automated judges. GitHub stars: 235."
   },
   {
    "name": "Pelican-Sim 1.0 policy evaluation and ranking (RoboTwin)",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "Beijing Innovation Center of Humanoid Robotics (X-Humanoid), WFM System Group",
    "date": "2026-09-10",
    "measures": "Whether the Pelican-Sim world model combined with a fine-tuned VLM judge reproduces RoboTwin simulator success rates and the ranking of five checkpoints from one VLA training run.",
    "scoring": "Pearson r, Spearman rho, Kendall tau, MMRV and MAE (percentage points) between the world-model-plus-VLM success estimate and the RoboTwin task-checker success, on 200 held-out initial conditions per checkpoint, at five world-model adaptation budgets (0, 100, 200, 500, 1,000 task-specific RoboTwin rollouts). The same policy-predicted action trajectory is executed in RoboTwin and supplied to the world model (matched-action, not closed-loop). Judge: fine-tuned Qwen3-VL-2B-Instruct.",
    "licence": "Code: Apache-2.0 (github.com/ZouShilong1024/Pelican-Sim1.0, linked from the project page; on 2026-10-11 the repo held only a README); Data: unknown (looked at: paper, project page, repo)",
    "url": "https://arxiv.org/abs/2609.12036",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2609.12036",
      "title": "Pelican-Sim 1.0: A General World Model Simulator for Embodied Intelligence (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://zoushilong1024.github.io/Pelican-Sim1.0/",
      "title": "Pelican-Sim 1.0 project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/ZouShilong1024/Pelican-Sim1.0",
      "title": "ZouShilong1024/Pelican-Sim1.0 repository (GitHub API licence: Apache-2.0; README only)",
      "type": "repo",
      "date": "2026-09-11",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Comparison target is a simulator (RoboTwin), not real robots. The paper header names GitHub 'Open-X-Humanoid/Pelican-Sim1.0', which returned 404 via the GitHub API on 2026-10-11. Technical report; the world model uses a 28-dimensional unified action space, URDF-rendered action videos and sparse MoE layers."
   },
   {
    "name": "DexTouch-WM world models as policy evaluators",
    "kind": "protocol",
    "domains": [
     "robotics"
    ],
    "org": "HKUST (Guangzhou); Xspark AI; Peking University; Tsinghua University; University of Hong Kong",
    "date": "2026-09-17",
    "measures": "Whether scores of three VLA policies rolled out closed-loop inside two task-adapted visuo-tactile world models match their scores on a real dexterous-hand robot on four tasks.",
    "scoring": "Per-task Pearson correlation and MMRV over N = 3 policies between imagined and real scores. Task rubrics map to [0,1]; imagined rollouts lose 0.5 points for clear physical inconsistencies. 10 matched rollouts per policy-task pair per environment, each scored by five human raters.",
    "licence": "Code: unknown; Data: unknown (looked at: arXiv abstract page and full text; no project page or repository link found)",
    "url": "https://arxiv.org/abs/2609.20649",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2609.20649",
      "title": "DexTouch-WM: Learning Action-Conditioned Tactile World Models from Human Touch for Dexterous Robot Manipulation (arXiv abstract; v1 2026-09-17, v2 2026-09-18)",
      "type": "paper",
      "date": "2026-09-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2609.20649",
      "title": "DexTouch-WM full text (arXiv HTML v2), Sec. IV-D, Tables II-III",
      "type": "paper",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Accepted to the IROS 2026 Workshop RoBoWoMo (lightning talk) per arXiv comments. Robot: Tianji arm with a 20-DoF Wuji dexterous hand. WM-Robot adapts on 200 robot trajectories; WM-Mix adapts on 100 robot and 100 human trajectories."
   },
   {
    "name": "VBench (including VBench++)",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "S-Lab, Nanyang Technological University; Shanghai Artificial Intelligence Laboratory; Nanjing University; The Chinese University of Hong Kong",
    "date": "2023-11-29",
    "measures": "General video generation quality split into 16 dimensions. Video Quality: subject consistency, background consistency, temporal flickering, motion smoothness, dynamic degree, aesthetic quality, imaging quality. Video-Condition Consistency: object class, multiple objects, human action, color, spatial relationship, scene, appearance style, temporal style, overall consistency. VBench++ adds image-to-video evaluation with an Image Suite, long-video evaluation, and trustworthiness (culture fairness, human bias, safety). There is no physics dimension.",
    "scoring": "Each dimension has its own automatic method (for example DINO and CLIP feature similarity, RAFT optical flow, frame-interpolation motion priors, LAION aesthetic predictor, MUSIQ, GRiT detection, UMT action recognition, Tag2Text, ViCLIP) on about 100 prompts per dimension. Leaderboard Total Score: each dimension is min-max normalised; Quality Score and Semantic Score are weighted averages (weight 1 per dimension, 0.5 for dynamic degree); Total Score = weighted average with Quality weight 4 and Semantic weight 1 (scripts/constant.py). Human validation: pairwise preferences per dimension (N prompts x 5 groups x 6 pairs), converted to model win ratios and correlated with VBench win ratios.",
    "licence": "Code: Apache-2.0 (LICENSE in github.com/Vchitect/VBench). Data: prompt suites ship in the same repo; no separate licence found for the sampled videos or the human preference annotations (looked at the repo LICENSE and README).",
    "url": "https://arxiv.org/abs/2311.17982",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2311.17982",
      "title": "VBench: Comprehensive Benchmark Suite for Video Generative Models (arXiv abstract page)",
      "type": "paper",
      "date": "2023-11-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2311.17982",
      "title": "VBench (arXiv HTML v1)",
      "type": "paper",
      "date": "2023-11-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2411.13503",
      "title": "VBench++: Comprehensive and Versatile Benchmark Suite for Video Generative Models (arXiv abstract page)",
      "type": "paper",
      "date": "2024-11-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2411.13503",
      "title": "VBench++ (arXiv HTML v1)",
      "type": "paper",
      "date": "2024-11-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Vchitect/VBench",
      "title": "VBench repository README and LICENSE",
      "type": "repo",
      "date": "2026-08-21",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Vchitect/VBench/blob/master/scripts/constant.py",
      "title": "VBench score weights (scripts/constant.py)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/spaces/Vchitect/VBench_Leaderboard",
      "title": "VBench Leaderboard (Hugging Face Space; table values read from the app config)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 16 dimensions, about 100 prompts per dimension, prompt suites for 8 content categories; 4 models in the CVPR paper (LaVie, ModelScope, VideoCrafter, CogVideo); VBench++ reports 32 additional models on the leaderboard. Venues per repo README: VBench at CVPR 2024 (Highlight); VBench++ in IEEE TPAMI 2025. Leaderboard: yes; on 2026-10-10 the text-to-video table had 73 rows, top Total Score HiDream-O1-Video at 89.74% (dated 2026-08-07, evaluated by the VBench team); the image-to-video table (35 rows) was topped by DreamX-World-1.0 at 90.49% (sampled and evaluated by its own team). Rows list who sampled and who evaluated each model. 1,812 GitHub stars on 2026-10-10. The leaderboard data file (HF dataset Vchitect/vbench_leaderboard_submission) is private; values were read from the public Space configuration."
   },
   {
    "name": "VideoPhy",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "University of California Los Angeles; Google Research",
    "date": "2024-06-05",
    "measures": "Whether text-to-video outputs follow the caption (semantic adherence) and physical commonsense in everyday interactions between materials: solid-solid, solid-fluid and fluid-fluid, with captions marked easy or hard by graphics researchers.",
    "scoring": "Human raters (Amazon Mechanical Turk; 14 workers with high-school physics who passed a qualification test) give binary scores for semantic adherence (SA) and physical commonsense (PC); 3 raters per test video, majority vote. Main metric: percentage of test prompts with SA=1 and PC=1. Automatic judge VideoCon-Physics: the 7B VideoCon video-language model fine-tuned with LoRA on human labels from the training split; it outputs Yes/No for SA and PC, and the automatic leaderboard averages the SA and PC rates.",
    "licence": "Code: MIT (LICENSE in github.com/Hritikbansal/videophy). Data: MIT per Hugging Face metadata of videophysics/videophy_test_public and videophysics/videophy_train_public; paper Appendix B says captions and human annotations are MIT and generated videos follow each video model owner's terms.",
    "url": "https://arxiv.org/abs/2406.03520",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2406.03520",
      "title": "VideoPhy: Evaluating Physical Commonsense for Video Generation (arXiv abstract page)",
      "type": "paper",
      "date": "2024-06-05",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2406.03520",
      "title": "VideoPhy (arXiv HTML v2)",
      "type": "paper",
      "date": "2024-10-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Hritikbansal/videophy",
      "title": "VideoPhy repository README (human and automatic leaderboards)",
      "type": "repo",
      "date": "2026-01-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/videophysics/videophy_test_public",
      "title": "videophy_test_public dataset metadata",
      "type": "dataset-card",
      "date": "2024-06-05",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/videophysics/videophy_train_public",
      "title": "videophy_train_public dataset metadata",
      "type": "dataset-card",
      "date": "2024-06-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 688 human-verified captions (candidates generated by GPT-4), split 344 train / 344 test; 12 text-to-video models in paper v2; 24,500 human annotations on test videos and 12,000 on training videos (about $3,500 in total). Best human-rated model: CogVideoX-5B with SA=1 and PC=1 on 39.6% of test prompts (paper v1 reported Pika at 19.7% before CogVideoX was added). Venue: repo README says ICLR 2025. Leaderboard: yes, human and automatic tables in the README (https://github.com/Hritikbansal/videophy#leaderboard-); human top CogVideoX-5B 39.6; automatic top CogVideoX-5B, average 49, over 14 models. The README says 10 models in the human leaderboard while its table lists 12. Project page videophy.github.io not opened."
   },
   {
    "name": "PhyGenBench (with PhyGenEval)",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Shanghai Jiao Tong University; OpenGVLab, Shanghai AI Laboratory; The University of Hong Kong; The Chinese University of Hong Kong",
    "date": "2024-10-07",
    "measures": "Whether text-to-video outputs show the correct physical phenomenon for prompts that each target one physical law, in four domains: mechanics, optics, thermal and material properties. Semantic alignment with the prompt is scored separately.",
    "scoring": "PhyGenEval scores physical commonsense alignment (PCA) in three stages: (1) key physical phenomena detection on keyframes with VQAScore, using retrieval prompts and questions written by GPT-4o; (2) physics order verification with GPT-4o or LLaVA-Interleave on keyframes; (3) overall naturalness of the whole video with GPT-4o or InternVideo2, against a prompt-specific standard written by GPT-4o. Each stage is mapped to a four-point scale (0-3), averaged and floored; stages 2 and 3 ensemble the two model choices. Model results are reported as averages per domain on a 0-1 scale. Semantic alignment uses GPT-4o checks of objects and actions. Human check: 3 annotators score 0-3.",
    "licence": "Code: no LICENSE file in github.com/OpenGVLab/PhyGenBench (GitHub API license field null). Data: the prompts are in the same repo; no data licence found (looked at repo root, README and project page). The project page's CC BY-SA 4.0 notice covers the website template only.",
    "url": "https://phygenbench123.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2410.05363",
      "title": "Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation (arXiv abstract page)",
      "type": "paper",
      "date": "2024-10-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2410.05363",
      "title": "PhyGenBench paper (arXiv HTML v1)",
      "type": "paper",
      "date": "2024-10-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://phygenbench123.github.io/",
      "title": "PhyGenBench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/OpenGVLab/PhyGenBench",
      "title": "PhyGenBench repository (README leaderboard, description)",
      "type": "repo",
      "date": "2024-10-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 160 prompts, 27 physical laws, 4 domains; 8 models (5 open, 3 closed) and 1,280 videos; prompts contain 165 unique objects and 42 unique actions (project page). Top model: Gen-3 with average PCA 0.51 (human score 0.48); Kling 0.49. Venue: the repo description says ICML 2025. Leaderboard: static tables on the project page ('Quantitative Evaluation') and in the README, 8 models. The README table differs slightly from paper Table 2 (CogVideoX-2B average 0.37 vs 0.39; CogVideoX-5B mechanics 0.43 vs 0.39; Lavie mechanics 0.40 vs 0.30)."
   },
   {
    "name": "Physics-IQ",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Google DeepMind; INSAIT, Sofia University (first author, work done while at Google DeepMind)",
    "date": "2025-01-14",
    "measures": "Whether a video generator can predict how a real, filmed physical experiment continues. Scenarios cover solid mechanics, fluid dynamics, optics, thermodynamics and magnetism. Each 8-second real video is split into a 3-second conditioning part and a 5-second test part. Video-to-video models get the 3 seconds; image-to-video models get only the last conditioning frame (the 'switch frame'); both can also get a text description that does not reveal the outcome.",
    "scoring": "The generated 5 seconds are compared with the real recording using four metrics on motion masks (thresholded frame differences): Spatial IoU (where motion happens), Spatiotemporal IoU (where and when), Weighted spatial IoU (where and how much) and MSE (pixel error). The four values are summed (MSE with a negative sign) into the Physics-IQ score, normalised so that two real takes of the same experiment ('physical variance') score 100%. No human raters and no judge model are used for the physics score. Visual realism is measured separately: Gemini 1.5 Pro sees a real and a generated video of the same scenario and must pick the generated one (two-alternative forced choice, chance 50%).",
    "licence": "Code: Apache-2.0 for all software (LICENSE file of github.com/google-deepmind/physics-IQ-benchmark; the GitHub API reports NOASSERTION because the file is a combined notice). Data: CC-BY 4.0 for all other materials (same LICENSE file and README licence section).",
    "url": "https://physics-iq.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2501.09038",
      "title": "Do generative video models understand physical principles? (arXiv abstract page)",
      "type": "paper",
      "date": "2025-01-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2501.09038",
      "title": "Do generative video models understand physical principles? (arXiv HTML v3)",
      "type": "paper",
      "date": "2025-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://physics-iq.github.io/",
      "title": "Physics-IQ Benchmark project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/google-deepmind/physics-IQ-benchmark/blob/main/LICENSE",
      "title": "Physics-IQ LICENSE",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/google-deepmind/physics-IQ-benchmark#leaderboard",
      "title": "Physics-IQ and Physics-IQ Verified leaderboards (repo README)",
      "type": "leaderboard",
      "date": "2026-10-08",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 66 scenarios, each filmed from 3 views (left, center, right) and recorded twice, giving 396 videos (66 x 3 x 2) at 3840x2160 and 30 FPS; the 198 take-1 videos are the generation inputs and the second takes give the physical-variance ceiling. The paper evaluates 8 model variants (VideoPoet i2v and multiframe, Lumiere i2v and multiframe, Runway Gen 3, Pika 1.0, Stable Video Diffusion, Sora); best was VideoPoet (multiframe) at 29.5%, Sora (i2v) scored 10.0%. Venue: the project page BibTeX lists WACV 2026 (Physics-IQ Verified cites it as WACV, pp. 948-958). Leaderboard: yes, in the repo README and rendered on the project page; the original-benchmark top entry is Magi-1 + GeoPhys (best-of-N) at 64.5% (video-to-video, added 2026-06-17); the top image-to-video entry is Cosmos3-Super + WMReward (best-of-N) at 48.9% (added 2026-05-26). The repo now calls Physics-IQ Verified the recommended variant. The project page links an ICCV 2025 challenge (not opened). 369 GitHub stars on 2026-10-10."
   },
   {
    "name": "VideoPhy-2",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "University of California Los Angeles; Google Research",
    "date": "2025-03-09",
    "measures": "Physical commonsense and prompt adherence of text-to-video outputs for real-world actions (sports and physical activities, object interactions), plus whether specific physical rules for each video are followed.",
    "scoring": "Human raters (12 qualified Amazon Mechanical Turk annotators, 3 per video) rate semantic adherence (SA) and physical commonsense (PC) on a 1-5 scale and mark candidate physical rules as followed, violated or cannot be determined; SA and PC are averaged and rounded, rules by majority vote. Candidate rules are generated by Gemini-2.0-Flash-Exp from video captions. Main metric: joint score = share of videos with SA >= 4 and PC >= 4. Automatic judge VideoPhy-2-Autoeval: a 7B model made by fine-tuning VideoCon-Physics on about 50K human annotations; outputs SA (1-5), PC (1-5) and rule labels (0 violated, 1 followed, 2 cannot be determined).",
    "licence": "Code: MIT (LICENSE in github.com/Hritikbansal/videophy, VIDEOPHY2 folder). Data: MIT per Hugging Face metadata of videophysics/videophy2_test and videophysics/videophy2_train.",
    "url": "https://videophy2.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.06800",
      "title": "VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation (arXiv abstract page)",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.06800",
      "title": "VideoPhy-2 (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://videophy2.github.io/",
      "title": "VideoPhy-2 project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Hritikbansal/videophy/tree/main/VIDEOPHY2",
      "title": "VideoPhy-2 README and human leaderboard",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/videophysics/videophy2_test",
      "title": "videophy2_test dataset metadata",
      "type": "dataset-card",
      "date": "2025-03-08",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/videophysics/videophy2_train",
      "title": "videophy2_train dataset metadata",
      "type": "dataset-card",
      "date": "2025-03-08",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Scale: 3,940 prompts from 197 seed actions (the abstract and project page say 200 actions); test split 590 prompts (3 per action), training split 3,350; hard subset of 60 actions (1,200 prompts) where CogVideoX-5B had zero joint score. 7 models evaluated; Sora only on 60 prompts through its web interface, Ray2 on 394 videos. Benchmark annotations: 10.2K SA, 10.2K PC and 30.6K rule labels (about $2,600); training annotations about 50K (about $3,515). Results: Wan2.1-14B best with joint 32.6% (all) and 21.9% (hard); conservation of momentum and of mass are the most violated laws (violation score 40%). Venue: repo README says ICLR 2026. Leaderboard: yes, human leaderboard in VIDEOPHY2/README.md, top Wan2.1-T2V-14B (32.6 all, 21.9 hard). The auto-evaluator is named VideoPhy-2-Autoeval in the paper body, VideoPhy-AutoEval in the abstract and VideoPhy2-eval on the project page."
   },
   {
    "name": "IPV-Bench (Impossible Videos)",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Show Lab, National University of Singapore",
    "date": "2025-03-18",
    "measures": "Whether video generators can follow prompts for impossible events that defy physical, biological, geographical or social laws while keeping high visual quality; separately, whether Video-LLMs understand impossible videos.",
    "scoring": "Generation: human annotators give binary labels for visual quality and for impossible-prompt following; IPV-Score is the share of videos that pass both. Automatic surrogate: a weighted combination of six VBench factors for visual quality and a three-step GPT-4o yes/no judgment for prompt following, multiplied after rescaling.",
    "licence": "Data: MIT per Hugging Face metadata of showlab/ImpossibleVideos. Code: GitHub API reports no licence for showlab/Impossible-Videos (repository not checked against the paper's links). The arXiv text itself is CC BY-NC-SA 4.0.",
    "url": "https://arxiv.org/abs/2503.14378",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.14378",
      "title": "Impossible Videos (arXiv abstract page)",
      "type": "paper",
      "date": "2025-03-18",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2503.14378",
      "title": "Impossible Videos (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-03-18",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://proceedings.mlr.press/v267/bai25a.html",
      "title": "Impossible Videos, Proceedings of the 42nd International Conference on Machine Learning (PMLR 267)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/showlab/ImpossibleVideos",
      "title": "ImpossibleVideos dataset metadata",
      "type": "dataset-card",
      "date": "2025-03-21",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Taxonomy: 4 domains, 14 categories; IPV-Txt has 260 prompts (2,600 generated videos for 10 models); IPV-Vid has 902 videos for the understanding task. 10 generation models; top Mochi 1 with IPV-Score 37.3%. Venue: ICML 2025 (PMLR v267). Not physics-only: it rewards following counterfactual prompts. arXiv ID found with a web search."
   },
   {
    "name": "VBench-2.0",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Shanghai Artificial Intelligence Laboratory; S-Lab, Nanyang Technological University; Sun Yat-sen University; The Chinese University of Hong Kong",
    "date": "2025-03-27",
    "measures": "'Intrinsic faithfulness' of generated video in five dimensions with 18 capabilities: Human Fidelity (human anatomy, identity, clothes); Creativity (diversity, composition); Controllability (dynamic spatial relationship, dynamic attribute, motion order understanding, human interaction, complex landscape, complex plot, camera motion); Physics (State Change: mechanics, thermotics, material; Geometry: multi-view consistency); Commonsense (motion rationality, instance preservation).",
    "scoring": "About 70 prompts per capability. Generalist judges: VLMs answering questions prepared with GPT-4o (for example whether the expected mechanical, thermal or material change is visible), or a VLM description followed by an LLM yes/no judgment. Specialist models: anomaly detectors trained for human anatomy and instance preservation; SIFT and FLANN feature matching with RANSAC, plus RAFT camera-speed correction, for multi-view consistency. Scores are 0-1 per capability. Human validation: pairwise preferences (group comparisons for diversity) on four models, 284 hours across 18 annotators; per-dimension correlation of VBench-2.0 win ratios with human win ratios.",
    "licence": "Code: Apache-2.0 (root LICENSE of github.com/Vchitect/VBench; the VBench-2.0 folder has no separate licence file). Data: prompts are in the repo; no separate data licence found.",
    "url": "https://arxiv.org/abs/2503.21755",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2503.21755",
      "title": "VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness (arXiv abstract page)",
      "type": "paper",
      "date": "2025-03-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.21755",
      "title": "VBench-2.0 (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-08-20",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Vchitect/VBench/tree/master/VBench-2.0",
      "title": "VBench-2.0 code folder",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/spaces/Vchitect/VBench_Leaderboard",
      "title": "VBench Leaderboard, VBench-2.0 table",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Authors' stated reason: VBench-style metrics (per-frame aesthetics, temporal consistency, basic prompt adherence) measure 'superficial faithfulness'; recent models score well on them but still break physical laws, commonsense, anatomy and composition, so VBench-2.0 targets 'intrinsic faithfulness'. The paper evaluates 4 models (HunyuanVideo, CogVideoX-1.5, Sora-480p, Kling 1.6) with a shared prompt refiner for the three non-Sora models. Leaderboard: yes; the VBench-2.0 table had 12 rows on 2026-10-10; top Total Score Veo 3 at 66.72% (Physics Score 69.35%, dated 2025-09-04, evaluated by the VBench team); highest Physics Score JT-CV at 75.23% (sampled and evaluated by its own team, 2026-04-01). No venue shown on arXiv or in the README."
   },
   {
    "name": "WorldScore",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "3d"
    ],
    "org": "Stanford University",
    "date": "2025-04-01",
    "measures": "World generation from an image plus a camera-trajectory layout, posed as a sequence of next-scene tasks, so that 3D scene generators, 4D generators and video generators can be compared. Three aspects: controllability, quality and dynamics, over static worlds (indoor and outdoor) and dynamic worlds (5 motion types).",
    "scoring": "Ten automatic metrics: camera control (rotation and translation error against the instructed trajectory), object control (open-set detection success), content alignment (CLIPScore), 3D consistency (DROID-SLAM reprojection error), photometric consistency (optical-flow end-point error), style consistency (Gram-matrix difference between first and last frame), subjective quality (mean of CLIP-IQA+ and CLIP Aesthetic, chosen with a 400-participant study), motion accuracy, motion magnitude and motion smoothness. Each is linearly normalised to 0-100 with empirical bounds. WorldScore-Static = mean of the controllability and quality metrics; WorldScore-Dynamic also averages in the three dynamics metrics (3D models get 0 on dynamics). No physics-correctness metric.",
    "licence": "Code: MIT (LICENSE in github.com/haoyi-duan/WorldScore). Data: MIT per Hugging Face metadata of Howieeeee/WorldScore.",
    "url": "https://haoyi-duan.github.io/WorldScore/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2504.00983",
      "title": "WorldScore: A Unified Evaluation Benchmark for World Generation (arXiv abstract page)",
      "type": "paper",
      "date": "2025-04-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2504.00983",
      "title": "WorldScore (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-11-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://haoyi-duan.github.io/WorldScore/",
      "title": "WorldScore project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/spaces/Howieeeee/WorldScore_Leaderboard",
      "title": "WorldScore Leaderboard (leaderboard.csv)",
      "type": "leaderboard",
      "date": "2026-09-10",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/haoyi-duan/WorldScore",
      "title": "WorldScore repository (MIT LICENSE)",
      "type": "repo",
      "date": "2026-07-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/Howieeeee/WorldScore",
      "title": "WorldScore dataset metadata",
      "type": "dataset-card",
      "date": "2025-03-26",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 3,000 test examples (2,000 static, 1,000 dynamic). Models: the abstract (v2) says 19; the body and Table 2 say 20 (13 video, 6 3D scene, 1 4D). Paper results: WonderWorld (72.69) and LucidDreamer (70.40) lead WorldScore-Static; CogVideoX-I2V is the best video model (62.15 static, 59.12 dynamic). Venue: ICCV 2025 (arXiv comments). Leaderboard: yes, 35 rows on 2026-10-11 (last modified 2026-09-10); top WorldScore-Static UniWorld-View at 85.53 (2026.07.23) and top WorldScore-Dynamic WorldScape-0.2(MoE) at 76.23 (2026.07.13), both sampled and evaluated by their own teams; among the 21 rows evaluated by the WorldScore team, the tops are WonderWorld 72.69 (static) and CogVideoX-I2V 59.12 (dynamic). Targets 3D, 4D and video world generators; it measures layout control and visual consistency, not physical correctness."
   },
   {
    "name": "Morpheus",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "University of Amsterdam; University of Trento",
    "date": "2025-04-03",
    "measures": "Whether video generators conditioned on frames of real, filmed Newtonian rigid-body experiments continue them in line with the governing equations of motion and with conservation of energy, momentum and period.",
    "scoring": "Objects are segmented and tracked to extract trajectories; physics-informed neural networks fit the known differential equation of each setup; the total score combines a dynamical score (fit to the governing dynamics) and a physical-invariance score (conservation). An automatic filter discards failed generations. One split uses Cosmos-Transfer1 style augmentation of the real first frames.",
    "licence": "Code: not checked (no repository link found in the text searched). Data: MIT per Hugging Face metadata of physics-from-video/morpheus-real-world (created 2026-07-04).",
    "url": "https://arxiv.org/abs/2504.02918",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2504.02918",
      "title": "Evaluating Newtonian Mechanics in Video Generative Models with Real Physical Systems (arXiv abstract page, v3)",
      "type": "paper",
      "date": "2025-04-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2504.02918v1",
      "title": "Morpheus: Benchmarking Physical Reasoning of Video Generative Models with Real Physical Experiments (arXiv v1 abstract)",
      "type": "paper",
      "date": "2025-04-03",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2504.02918",
      "title": "Morpheus (arXiv HTML v3)",
      "type": "paper",
      "date": "2026-06-29",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/physics-from-video/morpheus-real-world",
      "title": "morpheus-real-world dataset metadata",
      "type": "dataset-card",
      "date": "2026-07-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Title changed between versions: v1 was 'Morpheus: Benchmarking Physical Reasoning of Video Generative Models with Real Physical Experiments' and reported 80 real-world videos; v3 (2026-06-29) is 'Evaluating Newtonian Mechanics in Video Generative Models with Real Physical Systems' and reports 130 videos from about 130 experiments repeated 10-20 times. Models include Veo3, Kling, Wan2.1, Cosmos, CogVideoX, PyramidalFlow and LTX-Video (nine models in the VLM-judge comparison). Venue: ICML 2026 (arXiv comments). No leaderboard found. arXiv ID found with a web search."
   },
   {
    "name": "T2VPhysBench",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Guilin University of Electronic Technology; University of Arizona; University of Wisconsin-Madison; Simons Institute for the Theory of Computing, UC Berkeley; Arizona State University",
    "date": "2025-05-01",
    "measures": "Whether text-to-video systems obey 12 physical laws in three groups: Newton's three laws and gravitation; conservation of energy, mass, linear and angular momentum; and phenomenological laws (Hooke's law, Snell's law, law of reflection, Bernoulli's principle). Also tests prompts with added law-specific hints and counterfactual prompts that ask for physics-breaking videos.",
    "scoring": "Fully manual: three student annotators rate every video on four levels mapped to 0.0, 0.25, 0.5 and 1.0; scores are averaged over prompts and annotators per model. No automatic judge.",
    "licence": "unknown (no code or data link found in the arXiv HTML; looked at the paper text and appendices list)",
    "url": "https://arxiv.org/abs/2505.00337",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2505.00337",
      "title": "T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation (arXiv abstract page)",
      "type": "paper",
      "date": "2025-05-01",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2505.00337",
      "title": "T2VPhysBench (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-05-01",
      "accessed": "2026-10-11"
     }
    ],
    "note": "10 models (closed and open); every model averages below 0.60 in every law category; best Wan 2.1 average 0.42, lowest SD Video 0.19; conservation laws score lowest. Number of prompts not found in the text read. No inter-annotator agreement reported (searched the paper text). No venue on arXiv. arXiv ID found with a web search."
   },
   {
    "name": "IntPhys 2",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "research"
    ],
    "org": "FAIR at Meta",
    "date": "2025-06",
    "measures": "Whether video models can tell physically possible from impossible events in synthetic videos.",
    "scoring": "Accuracy on possible/impossible video pairs, compared with human accuracy.",
    "licence": "Code and data: custom CC-BY-NC-4.0 with an evaluation-only restriction (LICENSE.md, HF card)",
    "url": "https://arxiv.org/abs/2506.09849",
    "in_atlas": true,
    "atlas_id": "intphys-2",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09849",
      "title": "IntPhys 2",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'intphys-2' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "MVPBench (Minimal Video Pairs)",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "research"
    ],
    "org": "FAIR at Meta",
    "date": "2025-06",
    "measures": "Physical understanding of video-language models, with near-identical video pairs that have opposite answers.",
    "scoring": "Paired accuracy on multiple-choice questions; a pair counts only if both are answered correctly.",
    "licence": "Code: CC-BY-NC-4.0 (repo LICENSE); Data: HF card says Apache-2.0 (conflict recorded in atlas); videos fetched from 9 original sources",
    "url": "https://arxiv.org/abs/2506.09987",
    "in_atlas": true,
    "atlas_id": "mvpbench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09987",
      "title": "A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'mvpbench' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "CausalVQA",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "research"
    ],
    "org": "FAIR at Meta",
    "date": "2025-06",
    "measures": "Cause-and-effect reasoning of video-language models on real Ego-Exo4D clips: counterfactual, anticipation and planning questions.",
    "scoring": "Paired accuracy on question pairs, compared with human accuracy.",
    "licence": "Code and data: Ego-Exo4D licence agreement (LICENSE.pdf in repo); no separate code licence",
    "url": "https://arxiv.org/abs/2506.09943",
    "in_atlas": true,
    "atlas_id": "causalvqa",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.09943",
      "title": "CausalVQA",
      "type": "paper",
      "date": "2025-06-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Details, scores and caveats are in the atlas entry 'causalvqa' (checked 2026-10-10); this row only places it in the map."
   },
   {
    "name": "WorldPrediction",
    "kind": "benchmark",
    "domains": [
     "general-video",
     "research"
    ],
    "org": "Meta FAIR Paris; The Hong Kong University of Science and Technology; ISIR Sorbonne Université",
    "date": "2025-06-04",
    "measures": "High-level world modeling and long-horizon procedural planning on human activity video. WorldPrediction-WM: given an initial and a final state image, pick the action clip that causes the change. WorldPrediction-PP: pick the correctly ordered sequence of 3-10 action clips. Candidate actions are shown in other scenes ('action equivalents') so background continuity cannot be used.",
    "scoring": "Multiple choice among 4 candidates (chance 25%); the metric is accuracy. It accepts VLMs, Socratic LLMs (caption first, then reason in text), video diffusion models (generate a continuation for each candidate action caption and pick the one whose last frame is closest in pixels to the final state) and trained procedural planners.",
    "licence": "Code and annotations: CC BY-NC 4.0 (LICENSE file in github.com/facebookresearch/WorldPrediction; the README says the same; GitHub API shows NOASSERTION). Source videos come from COIN, CrossTask, EgoExo4D, EPIC-KITCHENS-100 and IKEA-ASM and must be obtained under their own terms.",
    "url": "https://arxiv.org/abs/2506.04363",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.04363",
      "title": "WorldPrediction: A Benchmark for High-level World Modeling and Long-horizon Procedural Planning (arXiv abstract page)",
      "type": "paper",
      "date": "2025-06-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.04363",
      "title": "WorldPrediction (arXiv HTML v1)",
      "type": "paper",
      "date": "2025-06-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/facebookresearch/WorldPrediction",
      "title": "WorldPrediction repository README and LICENSE",
      "type": "repo",
      "date": "2025-12-17",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 825 WM samples and 570 PP samples, kept from 1,500 candidates per task after human filtering; 1,800 and 749 unique actions. Best results: Qwen2.5-VL 72B at 57.0% on WM; Claude-3.5-sonnet (Socratic LLM) at 38.1% on PP; video diffusion models CogVideoX-I2V 30.1% and I2VGenXL 26.1% on WM. Human accuracy is perfect by construction: only samples that both annotators answered correctly were kept (inter-annotator agreement before filtering 0.73 for WM, 0.65 for PP; 34 and 46 annotators). Section 4.3 says the best WM model reaches 45% (Claude-3.5), which conflicts with Table 2 (Claude-3.5-sonnet 53.3%, Qwen2.5-VL 72B 57.0%). No leaderboard found (looked at the repo README). No venue on arXiv or in the README. This is a discriminative test: it does not score generated video quality."
   },
   {
    "name": "PhyWorldBench",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "University of California, Santa Cruz; NVIDIA Research; Northeastern University; University of California, Santa Barbara",
    "date": "2025-07-17",
    "measures": "Physical realism of text-to-video output across 10 physics categories, from object motion and energy conservation to rigid-body interactions and human or animal motion, plus an Anti-Physics category whose prompts ask for physics-violating events.",
    "scoring": "Human raters (Amazon Mechanical Turk, 3 per video, majority vote) answer yes/no on object presence, event presence and visibility of the expected physical phenomena. Semantic adherence (SA) requires the objects and the event; physical commonsense (PC) requires all 'Key Standards'; success requires both. Automatic judge: Context-Aware Prompt (CAP), a zero-shot multimodal LLM (GPT-o1 in the paper) told that the video is AI-generated and asked to describe the video before giving a yes/no answer.",
    "licence": "Code: github.com/ashwin-333/phy-world-bench has no licence (GitHub API null). Data: generated videos and prompt JSON on Hugging Face phyworldbench/phyworldbench, MIT per its metadata; the paper says those generated videos are not part of the benchmark itself.",
    "url": "https://arxiv.org/abs/2507.13428",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2507.13428",
      "title": "PhyWorldBench: A Comprehensive Evaluation of Physical Realism in Text-to-Video Models (arXiv abstract page)",
      "type": "paper",
      "date": "2025-07-17",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2507.13428",
      "title": "PhyWorldBench (arXiv HTML, latest version)",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/79d86433c2acd12b6fa98553435d226e-Abstract-Conference.html",
      "title": "PhyWorldBench in ICLR 2026 proceedings",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/api/datasets/phyworldbench/phyworldbench",
      "title": "phyworldbench dataset metadata",
      "type": "dataset-card",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 1,050 prompts (10 categories x 5 subcategories x 7 scenarios x 3 prompt types); 12 models and 12,600 videos. The body says 5 proprietary and 7 open-source models; the abstract says 'five open-source and five proprietary'. Best: Pika 2.0 with success rate 0.262. Venue: ICLR 2026 (proceedings page). Leaderboard: only Appendix P, Table 15 (human and CAP rankings); no online leaderboard found. arXiv ID found with a web search."
   },
   {
    "name": "WorldMark",
    "kind": "benchmark",
    "domains": [
     "games",
     "general-video"
    ],
    "org": "Alaya Lab; The University of Tokyo; Shanghai Innovation Institute",
    "date": "2026-04-23",
    "measures": "Interactive image-to-video world models driven by navigation commands: how motion reacts to each command (direction, purity, latency, stability, per translation and rotation axis), whether the generated world stays the same over time (world memory at three time scales), and visual quality, across first- and third-person views, real and stylized scenes, and three difficulty tiers.",
    "scoring": "v2: per-model adapters translate one WASD-style action vocabulary into each model's native control format (captions, 6-DoF poses, camera trajectories, action functions). Nine deterministic metrics computed on generated video only, normalized to 0 to 100: Direction Accuracy, Direction Purity, Response Latency and Motion Stability (each per axis); Local, Global and Revisit Memory; Visual Quality (an existing perceptual scorer). 500 standardized cases; 125 videos per model per split. v1 (2026-04-23) used Visual Quality, Control Alignment and World Consistency metrics, partly scored by a VLM (Gemini 3.1 Pro), on six models.",
    "licence": "Code: MIT (LICENSE in github.com/AlayaLab/WorldMark, read via gh api 2026-10-11). Data: test inputs are in the repository's arena_inputs folder and fall under the repository licence (inferred); the paper says all data, evaluation code and model outputs will be released.",
    "url": "https://arxiv.org/abs/2604.21686",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2604.21686",
      "title": "WorldMark: A Unified Benchmark Suite for Interactive Video World Models (arXiv abs)",
      "type": "paper",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2604.21686v2",
      "title": "WorldMark arXiv HTML full text v2 (Tables 3-4, Section 4.3, Appendix G)",
      "type": "paper",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2604.21686v1",
      "title": "WorldMark arXiv HTML full text v1",
      "type": "paper",
      "date": "2026-04-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/AlayaLab/WorldMark",
      "title": "AlayaLab/WorldMark GitHub repository",
      "type": "repo",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     }
    ],
    "note": "v1 2026-04-23 and v2 2026-08-05 differ substantially. v1 evaluated six models (Yume 1.5, Matrix-Game 2.0, HY-World 1.5, HY-GameCraft, Open-Oasis, Genie 3). v2 evaluates ten distilled few-step models (Yume 1.5, HY-World 1.5, HY-GameCraft 1.0, Matrix-Game 2.0, Matrix-Game 3.0, LingBot-World, SANA-WM, DreamX-World, AlayaWorld, Lyra 2.0) and excludes Genie 3 because it can only be operated manually through a web interface. AlayaWorld comes from Alaya Lab, and its reference in WorldMark lists authors who are also WorldMark authors (K. Zhang, Y. Ge, Y. Yin, K. He). The paper links a 'World Model Arena' site (warena.ai), not opened. Project page alayalab.github.io/WorldMark not opened. GitHub stars: 44 on 2026-10-11."
   },
   {
    "name": "WBench",
    "kind": "benchmark",
    "domains": [
     "games",
     "general-video"
    ],
    "org": "Fudan University; Meituan LongCat Team",
    "date": "2026-05-25",
    "measures": "Interactive video world models in multi-turn use: video quality, adherence to a stated world setting, adherence to interactions (navigation, subject action, event editing, perspective switching), consistency, and physics compliance.",
    "scoring": "22 automatic sub-metrics that combine specialist vision models with large multimodal models, grouped into five dimensions. 289 test cases with 1,058 interaction turns, first- and third-person. Navigation commands are given as text, relative 6-DoF poses or discrete keys depending on the model. Text-driven models run all 289 cases; camera-controlled and action-conditioned models run the 158-case navigation subset.",
    "licence": "Code: MIT (LICENSE in github.com/meituan-longcat/WBench, read via gh api 2026-10-11). Data: Hugging Face dataset meituan-longcat/WBench card states mit; the paper says test data are built from synthetic or openly licensed imagery.",
    "url": "https://arxiv.org/abs/2605.25874",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2605.25874",
      "title": "WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation (arXiv abs)",
      "type": "paper",
      "date": "2026-05-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2605.25874v1",
      "title": "WBench arXiv HTML full text v1 (Sections 5.1-5.4, Appendix D.2)",
      "type": "paper",
      "date": "2026-05-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/meituan-longcat/WBench",
      "title": "meituan-longcat/WBench GitHub repository (README news, LICENSE)",
      "type": "repo",
      "date": "2026-10-08",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/meituan-longcat/WBench",
      "title": "WBench dataset card",
      "type": "dataset-card",
      "date": "2026-05-29",
      "accessed": "2026-10-11"
     }
    ],
    "note": "20 models in the paper: 9 text-driven (e.g. Seedance 1.5, Kling 3.0, Wan 2.7), 5 camera-controlled (e.g. HY-World 1.5, LingBot-World), 6 action-conditioned (e.g. Genie 3, Matrix-Game 3.0). Leaderboard on the project page (meituan-longcat.github.io/WBench/#leaderboard, not opened); README news reports 29 models by 2026-07-29 and later entries up to 2026-10-04, some marked 'self-evaluation' (run by the submitting team). arXiv comment: technical report; no venue. GitHub stars: 246 on 2026-10-11."
   },
   {
    "name": "Human World Bench (HWB)",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "NVIDIA (Cosmos 3 technical report)",
    "date": "2026-06-01",
    "measures": "Egocentric image-to-video generation of human manipulation tasks under task-level instructions: whether the video shows the requested actions and objects, and whether motion, object dynamics, contact and hand anatomy are physically plausible.",
    "scoring": "Human annotators judge each generated video independently (absolute failure-mode protocol) and give pass/fail on instruction following and on physical plausibility; the HWB score is the average of the two pass rates. The report also uses HWB's action annotations (camera ego-motion and hand tracking) to score action-conditioned forward dynamics by PSNR.",
    "licence": "not released as far as found (looked at: Cosmos 3 report v4, whose 'Open Evaluation Benchmark' release list names only Cosmos-HUE at huggingface.co/datasets/nvidia/Cosmos-HumanEval-v1; HF dataset search for HumanWorldBench and Human-World-Bench, no results)",
    "url": "https://arxiv.org/abs/2606.02800",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.02800",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI (arXiv abs; v1 2026-06-01, v4 2026-06-23)",
      "type": "report",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3 report full text v4 (Sec. 6.2.2 Human World Bench, Table 14)",
      "type": "report",
      "date": "2026-06-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2606.02800v1",
      "title": "Cosmos 3 report full text v1 (HWB text and numbers already present)",
      "type": "report",
      "date": "2026-06-01",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Scale: 180 samples sourced from EgoVerse (cited in the report as an egocentric human dataset for robot learning). 9 models have HWB scores (Table 14): Cosmos3-Super 71.9, Veo-3.1 67.8, Cosmos3-Nano 66.9, Wan2.2-A14B 60.7, HunyuanVideo-1.5 54.7, Cosmos-Predict2.5-14B 38.7, Wan2.1-14B 33.1, Cosmos-Predict2.5-2B 32.8, Wan2.2-5B 25.4. The top two are NVIDIA's own models, scored by NVIDIA's protocol. Number of annotators and agreement are not stated. Forward-dynamics PSNR: Cosmos3-Super 16.19 dB (MT-init) vs LOME 9.36 dB. The HWB text and numbers are identical in v1 (2026-06-01) and v4 (2026-06-23). Leaderboard: none."
   },
   {
    "name": "Physics-IQ Verified",
    "kind": "benchmark",
    "domains": [
     "general-video"
    ],
    "org": "Anates Labs; Technical University of Munich; University of Technology Nuremberg; Tuebingen AI Center, University of Tuebingen; Helmholtz AI, Munich; Google DeepMind (Jaini and Geirhos, described as advisory in the acknowledgements)",
    "date": "2026-06-17",
    "measures": "The same task and the same 66 real-world scenarios as Physics-IQ, after an audit that corrected prompts and ground-truth motion maps, so that the score reflects physical prediction rather than prompt ambiguity or unrelated motion in the recordings.",
    "scoring": "Same four metrics as Physics-IQ (Spatial, Spatiotemporal and Weighted spatial IoU, MSE), computed against ground truth from which 'artifacts' (motion not caused by the physical effect) were removed with manual annotations. Each metric of each sample is divided by that sample's own physical variation and clipped to [0,1] (MSE as the inverse ratio); the four are averaged per sample and then over samples, so every sample and metric weighs the same. Prompts are split into six fields (SETUP, SCENE, ACTION, CAM, STYLE, SCOPE) and rendered by model-specific templaters ('best-practice prompts', bpp); the original prompts ('op') remain available. No human raters or judge models. The leaderboard recommends 4 runs (mean and standard deviation) and requires them to claim the top spot.",
    "licence": "Code: same repository as Physics-IQ (github.com/google-deepmind/physics-IQ-benchmark), whose LICENSE file puts all software under Apache-2.0. Data: CC-BY-4.0 per the Hugging Face metadata of Anates-Labs-Research/Physics-IQ-Verified (gated, approval automatic; the card text requires login and was not read). Leaderboard uploads fall under separate 'Physics-IQ Verified Submission Terms' linked in the README (not read).",
    "url": "https://arxiv.org/abs/2606.18943",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2606.18943",
      "title": "Physics-IQ Verified (arXiv abstract page)",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2606.18943v1",
      "title": "Physics-IQ Verified (arXiv PDF v1, affiliation block)",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/google-deepmind/physics-IQ-benchmark",
      "title": "Physics-IQ benchmark repository README (Verified workflow, leaderboard, disclaimer)",
      "type": "repo",
      "date": "2026-10-08",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/Anates-Labs-Research/Physics-IQ-Verified",
      "title": "Physics-IQ Verified dataset metadata (Hugging Face API)",
      "type": "dataset-card",
      "date": "2026-06-19",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://physics-iq-verified.anates.ai",
      "title": "Physics-IQ Verified leaderboard site",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The DeepMind repo README calls Physics-IQ Verified the 'recommended benchmark variant' and hosts its leaderboard. The same README states: 'Physics-IQ Verified is an independent third-party benchmark that is not endorsed or verified by Google DeepMind'. The paper evaluates six image-to-video models (Wan 2.2, HunyuanVideo 1.5, Cosmos3-Nano, Sora 2, P-Video, Grok Imagine Video), each with 4 complete runs of 198 videos per prompt setting, in a 2 x 2 x 2 design (prompts x ground truth x score). Leaderboard: yes, README and physics-iq-verified.anates.ai (the site showed '41 models 16 labs' on 2026-10-10). Top README entry: Odyssey-3 Pro + best-of-8 (WMReward + MBR consensus, custom prompts), 66.10, video-to-video, added 2026-10-08; top image-to-video entry: FLUX 3 [large] (Claude Opus 5.5 prompts) + WMReward + consensus (best-of-N), 54.70 ± 0.41, added 2026-10-06. Many top entries use best-of-N selection and custom prompts. The README warns that test data must not be used for best-of-N selection or prompt rewriting. No venue shown on arXiv."
   },
   {
    "name": "PlayWorld",
    "kind": "benchmark",
    "domains": [
     "games",
     "general-video"
    ],
    "org": "The Chinese University of Hong Kong; The University of Hong Kong; Zhejiang University; Kling Team, Kuaishou Technology (as listed in the arXiv HTML; the code repository now resolves to github.com/hku-sail/PlayWorld)",
    "date": "2026-08-13",
    "measures": "Long-horizon capability of interactive video world models when a player pursues a stated objective: geometry consistency, interaction fidelity, out-of-sight evolution (what happens to things while unseen) and insight evolution (processes that should change while watched), plus basic video quality and action controllability.",
    "scoring": "A multimodal Agent Player (Claude Haiku in the main setup) drives each world model toward the scenario objective, starting from a shared basic action sequence and adapting actions online to the model's control granularity; rollouts last about 10 to 60 s. Gemini 3.1 Pro answers sample-specific Yes/No rubric questions (more than 820 questions over 1,400+ videos) to give 1 to 5 scores per dimension; a validation gate (e.g. Trajectory Validity) sets the score to the minimum of 1 if the rollout never reached the objective region. Nine basic-ability metrics (VBench-style video quality, translation and rotation pass rates from VGGT camera poses) are combined by average rank into a Basic Ability Score.",
    "licence": "Code: no LICENSE file found (gh api license returns 404 for kxding/PlayWorld, which redirects to hku-sail/PlayWorld, checked 2026-10-11); the README says code and benchmark assets have different release considerations. Data: Hugging Face dataset jocelynd/playworld-bench has no licence field; the README says initial images come from Pexels and web image search and users must follow source-specific terms.",
    "url": "https://arxiv.org/abs/2608.13552",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2608.13552",
      "title": "PlayWorld: Benchmarking World Models with Agent Players over Long-Horizon Objectives (arXiv abs)",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2608.13552v2",
      "title": "PlayWorld arXiv HTML full text v2 (Tables 2, 3, 6; Figure 6)",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/kxding/PlayWorld",
      "title": "PlayWorld GitHub repository (redirects to hku-sail/PlayWorld)",
      "type": "repo",
      "date": "2026-08-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/jocelynd/playworld-bench",
      "title": "PlayWorld benchmark dataset (Hugging Face metadata)",
      "type": "dataset-card",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/spaces/jocelynd/PlayWorld-Leaderboard",
      "title": "PlayWorld leaderboard (Hugging Face Space, metadata via API)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "171 human-annotated scenarios. Nine models: Genie 3, LingBot-World, LingBot-World2, HY-World2, HappyOyster (through their web interfaces) and SANA-WM, Hunyuan-GameCraft, HY-WorldPlay, Matrix-Game-3.0 (run locally). Rubric results on a 1 to 5 scale (Table 2): Genie 3 overall 2.12 (first), HappyOyster 1.92, Matrix-Game-3.0 1.14 (last). arXiv v1 2026-08-13, v2 2026-08-14. Leaderboard: Hugging Face Space jocelynd/PlayWorld-Leaderboard (static, running). Repo stars: 92 on 2026-10-11. Name collision: not the same as PlayWorld (arXiv 2603.09030), a robot world model learned from autonomous play."
   },
   {
    "name": "DeepMind Control Suite",
    "kind": "benchmark",
    "domains": [
     "research"
    ],
    "org": "DeepMind (paper title and repository owner google-deepmind)",
    "date": "2018-01",
    "measures": "Continuous-control performance of reinforcement-learning agents on simulated MuJoCo bodies (cart-pole, cheetah, walker, humanoid, manipulator and others); model-based agents such as Dreamer report results on it.",
    "scoring": "Per-step rewards lie in [0, 1] for all domains except LQR, episodes last 1000 steps, so episode returns fall in [0, 1000].",
    "licence": "Code: Apache-2.0 (GitHub licence API, google-deepmind/dm_control)",
    "url": "https://arxiv.org/abs/1801.00690",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/1801.00690v1",
      "title": "DeepMind Control Suite",
      "type": "paper",
      "date": "2018-01-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/google-deepmind/dm_control/blob/main/LICENSE",
      "title": "dm_control LICENSE",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2301.04104v2",
      "title": "Mastering Diverse Domains through World Models (DreamerV3)",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Used by DreamerV3 as one of its 8 evaluation domains (Control Suite). The world model is judged through the agent's return."
   },
   {
    "name": "Atari 100k",
    "kind": "benchmark",
    "domains": [
     "games",
     "research"
    ],
    "org": "Introduced with SimPLe by Google Brain, deepsense.ai, Institute of Mathematics of the Polish Academy of Sciences, University of Warsaw, UIUC and Stanford (paper affiliations)",
    "date": "2019-03",
    "measures": "How well a reinforcement-learning agent plays Atari games when it may interact with each game only briefly; world-model agents (SimPLe, Dreamer and others) use it to show that learning inside a model saves real interaction.",
    "scoring": "Game score in each of 26 Atari games after 100K agent steps (400K frames, about two hours of play). The world model is judged only through the score of the agent trained with it.",
    "licence": "Protocol, no dataset. The underlying Arcade Learning Environment code is GPL-2.0 (GitHub licence API, Farama-Foundation/Arcade-Learning-Environment).",
    "url": "https://arxiv.org/abs/1903.00374",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/1903.00374v5",
      "title": "Model Based Reinforcement Learning for Atari (SimPLe)",
      "type": "paper",
      "date": "2019-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2301.04104v2",
      "title": "Mastering Diverse Domains through World Models (DreamerV3)",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Farama-Foundation/Arcade-Learning-Environment/blob/main/LICENSE.md",
      "title": "Arcade Learning Environment LICENSE",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The SimPLe paper evaluates at 100K interactions; DreamerV3 lists Atari100k as 26 games with a 400K-frame budget. Typical of how latent-dynamics world models are evaluated: by downstream control, with no separate test of prediction quality."
   },
   {
    "name": "Crafter",
    "kind": "benchmark",
    "domains": [
     "games",
     "research"
    ],
    "org": "Danijar Hafner (Google Research, Brain Team; University of Toronto)",
    "date": "2021-09",
    "measures": "Breadth of abilities of an agent in a 2D open-world survival game: exploration, resource collection, crafting, survival.",
    "scoring": "Success rate of each of 22 achievements over training episodes within a budget of 1M environment steps; the Crafter score is the geometric mean of the 22 success rates (in %).",
    "licence": "Code: MIT (GitHub licence API, danijar/crafter)",
    "url": "https://arxiv.org/abs/2109.06780",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2109.06780v2",
      "title": "Benchmarking the Spectrum of Agent Capabilities (Crafter)",
      "type": "paper",
      "date": "2021-09-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/danijar/crafter/blob/main/LICENSE",
      "title": "crafter LICENSE",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2301.04104v2",
      "title": "Mastering Diverse Domains through World Models (DreamerV3)",
      "type": "paper",
      "date": "2023-01-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": "ICLR 2022 (paper header). DreamerV3 uses Crafter in its scaling study."
   },
   {
    "name": "DriveArena",
    "kind": "benchmark",
    "domains": [
     "driving"
    ],
    "org": "Shanghai Artificial Intelligence Laboratory; Zhejiang University; Shanghai Jiao Tong University; East China Normal University; Technical University of Munich",
    "date": "2024-08-01",
    "measures": "Open-loop and closed-loop driving performance of camera-based driving agents inside a generative simulation platform. A Traffic Manager simulates traffic on nuScenes, CARLA or OpenStreetMap road networks, and World Dreamer (a layout-conditioned diffusion model trained on nuScenes) generates the six surround-view camera images the agent sees at each step.",
    "scoring": "Open-loop mode: PDM Score from NAVSIM (with changes: no at-fault distinction for collisions; the Traffic Manager's planned path is the progress reference), averaged over all frames. Closed-loop mode: Arena Driving Score ADS = route completion x PDMS; a run ends when the agent collides or leaves the road. World Dreamer fidelity is checked by running UniAD on generated versions of 150 nuScenes validation scenes (detection mAP/NDS, BEV segmentation mIoU, planning L2 and collision rate).",
    "licence": "Code: Apache-2.0 (LICENSE.txt in github.com/PJLab-ADG/DriveArena, read via gh api 2026-10-10). Weights: Hugging Face model jokester-yxm/DriveArena, linked from the official WorldDreamer README, card states apache-2.0. Data: no own dataset; World Dreamer is trained on nuScenes (v1.1 weights also on nuPlan), which carry their own terms.",
    "url": "https://pjlab-adg.github.io/DriveArena/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2408.00415",
      "title": "DriveArena: A Closed-loop Generative Simulation Platform for Autonomous Driving (arXiv abs)",
      "type": "paper",
      "date": "2024-08-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2408.00415v1",
      "title": "DriveArena arXiv PDF v1 (Tables 1 to 3)",
      "type": "paper",
      "date": "2024-08-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/PJLab-ADG/DriveArena",
      "title": "PJLab-ADG/DriveArena GitHub repository (README leaderboard, LICENSE)",
      "type": "repo",
      "date": "2025-09-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/jokester-yxm/DriveArena",
      "title": "DriveArena World Dreamer weights (Hugging Face model card metadata)",
      "type": "dataset-card",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://pjlab-adg.github.io/DriveArena/",
      "title": "DriveArena project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Only arXiv v1 exists. The paper evaluates one agent, UniAD: open-loop PDMS 0.910 on original nuScenes images, 0.902 on World Dreamer images of the same scenes, 0.636 in DriveArena's own open-loop simulation, and 0.950 for the human driving logs; closed loop on 4 routes (2 Boston, 2 Singapore, about 120 s each): PDMS 0.667, RC 0.137, ADS 0.086, which the paper calls preliminary. The README leaderboard repeats these UniAD numbers but lists ADS 0.1684 for sing_route_1, where the paper gives 0.1282; 0.7615 x 0.1684 = 0.1282, so the README value equals the RC value and appears to be an error (inferred). README news: V1.2 (2024-11-27) adds VAD support; no VAD scores in the README leaderboard. Venue: ICCV 2025 per the official GitHub description. Maps in the paper: singapore-onenorth, boston-seaport, boston-thomaspark, carla-town05. WorldLens later uses DriveArena maps and the ADS metric for its closed-loop tests. GitHub stars: 472 on 2026-10-10."
   },
   {
    "name": "ACT-Bench",
    "kind": "benchmark",
    "domains": [
     "driving"
    ],
    "org": "Turing Inc.",
    "date": "2024-12-06",
    "measures": "Action fidelity of driving world models: whether a model that generates front-camera driving video from context frames plus a commanded trajectory actually shows the commanded motion.",
    "scoring": "Each world model generates 2,286 videos, one per benchmark pair (3 nuScenes context frames plus an instruction trajectory from one of nine high-level action categories). ACT-Estimator (I3D backbone with self-attention, an action-classification head and a GRU trajectory head) estimates the executed action class and ego trajectory from each generated video. Instruction-Execution Consistency (IEC) is the share of videos whose estimated action class matches the instructed class. Trajectory Alignment is the ADE and FDE between the instructed and the estimated trajectory.",
    "licence": "Code: Apache-2.0 (LICENSE in github.com/turingmotors/ACT-Bench, read via gh api 2026-10-10). Data: Hugging Face dataset turing-motors/ACT-Bench card states apache-2.0; it stores trajectories and nuScenes file paths, and users must download nuScenes themselves under nuScenes terms. Models: ACT-Estimator and Terra are on Hugging Face with license tag cc-by-nc-sa-4.0 (Hugging Face API listing).",
    "url": "https://turingmotors.github.io/actbench/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2412.05337",
      "title": "ACT-Bench: Towards Action Controllable World Models for Autonomous Driving (arXiv abs)",
      "type": "paper",
      "date": "2024-12-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2412.05337",
      "title": "ACT-Bench arXiv HTML full text v1",
      "type": "paper",
      "date": "2024-12-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://turingmotors.github.io/actbench/",
      "title": "ACT-Bench project page (includes Terra v2 results)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/turingmotors/ACT-Bench",
      "title": "turingmotors/ACT-Bench GitHub repository",
      "type": "repo",
      "date": "2024-12-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/datasets/turing-motors/ACT-Bench",
      "title": "ACT-Bench dataset card",
      "type": "dataset-card",
      "date": "2024-12-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "2,286 video-trajectory pairs from the nuScenes validation split (CAM_FRONT only); per-category counts range from 89 (starting) to 541 (straight at constant speed). Models evaluated in the paper: Vista and Terra (Turing's own baseline, GAIA-1-style). Paper results: IEC 30.72% (Vista) vs 44.11% (Terra); average ADE 4.50 vs 3.98, FDE 8.66 vs 8.21. The project page adds Terra v2: accuracy 0.632, ADE 3.86, FDE 8.05. Only one arXiv version; no venue shown on arXiv or the repo. No leaderboard found. GitHub stars: 29 on 2026-10-10."
   },
   {
    "name": "Bench2Drive-R",
    "kind": "benchmark",
    "domains": [
     "driving"
    ],
    "org": "Shanghai Jiao Tong University (Dept. of CSE, School of AI, and MoE Key Lab of AI)",
    "date": "2024-12-11",
    "measures": "Reactive closed-loop evaluation of camera-based end-to-end driving models using real-world data: a rule-based behaviour controller built on the nuPlan simulator moves the other road users, and a diffusion-based generative renderer produces the camera images for each new state. Sensor rendering and behaviour control are separate modules. The paper also measures the renderer itself (image quality, layout adherence, temporal consistency).",
    "scoring": "Renderer: FID; BEVFormer and BEVFusion (camera branch) detection NDS/mAP and BEV segmentation mIoU on generated nuScenes images; StreamPETR scores for temporal consistency; UniAD open-loop planning L2 and collision rate. Closed loop: nuPlan Closed-Loop Score (reported as R-CLS) for a VAD planner on the Val14 split, using 10 clips from each of 14 scenario types, plus BEVFormer NDS/mAP on the rendered images.",
    "licence": "Code: unknown (the paper says the code will be open-sourced; no repository found via GitHub repository search for 'Bench2Drive-R' and github.com/Thinklab-SJTU/Bench2Drive-R returns 404, checked 2026-10-10). Data: no released data found; experiments use nuScenes and the nuPlan mini split.",
    "url": "https://arxiv.org/abs/2412.09647",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2412.09647",
      "title": "Bench2Drive-R: Turning Real World Data into Reactive Closed-Loop Autonomous Driving Benchmark by Generative Model (arXiv abs)",
      "type": "paper",
      "date": "2024-12-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2412.09647",
      "title": "Bench2Drive-R arXiv HTML full text v1",
      "type": "paper",
      "date": "2024-12-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2412.09647v1",
      "title": "Bench2Drive-R arXiv PDF v1 (Tables 2, 3, 6)",
      "type": "paper",
      "date": "2024-12-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Closed-loop result (Table 6): VAD R-CLS 30.49 with Bench2Drive-R rendering, 28.56 with static frame generation, 27.24 with log replay; BEVFormer NDS on the images 28.23, 23.31 and 0.05. Only one driving policy (VAD) is run in closed loop. The paper reports no comparison of its closed-loop scores with CARLA, NAVSIM or real-world driving outcomes. Renderer FID 10.95 vs 16.20 for a replicated MagicDrive. Only one arXiv version; no venue shown. Not to be confused with Bench2Drive (CARLA-based) or Bench2Drive-Robust."
   },
   {
    "name": "WorldLens",
    "kind": "benchmark",
    "domains": [
     "driving"
    ],
    "org": "WorldBench Team (the only group name on the arXiv paper, PDF title page and project page; no institutional affiliation block is given)",
    "date": "2025-12-11",
    "measures": "How well generative driving world models (multi-camera video generators conditioned on nuScenes scene layouts) produce video that looks real, keeps consistent 3D geometry, follows physics, lets a pretrained driving planner drive, and supports perception models trained on real data. Five aspects (Generation, Reconstruction, Action-Following, Downstream Task, Human Preference) with 24 dimensions in total.",
    "scoring": "One automatic metric per dimension. Generation (8): class-specific realism classifiers on object crops, ReID, DINO and CLIP feature similarity over time, depth-map smoothness, SegFormer label stability, FVD, LoFTR cross-view matching. Reconstruction (4): each generated video is lifted into a 4D Gaussian scene and scored with LPIPS/PSNR/SSIM, depth AbsRel against a reconstruction of the real video, MUSIQ and FVD on novel views. Action-Following (4): L2 distance between UniAD trajectories planned from generated and from real video; NAVSIM-style PDMS in open loop; Route Completion and Arena Driving Score (ADS = RC x PDMS) in closed loop inside a generative simulator built on DriveArena maps. Downstream Task (4): perception models pretrained on real nuScenes data run on generated video (BEV map mIoU, BEVFusion NDS, 3D tracking AMOTA, SparseOcc RayIoU). Human Preference: annotators rate each video 1 to 10 on World Realism (overall, vehicle, pedestrian), Physical Plausibility, 3D & 4D Consistency and Behavioral Safety. Results are reported per dimension next to an 'Empirical Max' reference; there is no single aggregate score. WorldLens-Agent (Qwen3-VL-8B fine-tuned with LoRA on 26,808 human scoring records) outputs a 1 to 10 score and a short rationale per human dimension.",
    "licence": "Code: Apache-2.0 (LICENSE in github.com/worldbench/WorldLens, read via gh api 2026-10-10; README says some bundled implementations carry other licences, see docs/LICENSE.md). Data: Hugging Face dataset worldbench/videogen (preparation files and generated videos) has a card stating apache-2.0; the underlying nuScenes data is listed in the paper's resource section as CC BY-NC-SA 4.0. WorldLens-26K and WorldLens-Agent are marked 'To be updated' in the README, their release is an unchecked TODO item, and no such dataset is listed under the worldbench Hugging Face account.",
    "url": "https://worldbench.github.io/worldlens",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.10958",
      "title": "WorldLens: Full-Spectrum Evaluations of Driving World Models in Real World (arXiv abs, v1 and v2)",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML full text v2",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2512.10958v2",
      "title": "WorldLens arXiv PDF v2 (title page, Table 2, Section 14)",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldbench.github.io/worldlens",
      "title": "WorldLens project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/worldbench/WorldLens",
      "title": "worldbench/WorldLens GitHub repository (README, LICENSE, repo metadata)",
      "type": "repo",
      "date": "2026-01-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/api/datasets/worldbench/videogen",
      "title": "Hugging Face dataset worldbench/videogen (card metadata)",
      "type": "dataset-card",
      "date": "2025-12-22",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://huggingface.co/spaces/worldbench/WorldLens",
      "title": "WorldLens leaderboard (Hugging Face Space, metadata via API)",
      "type": "leaderboard",
      "date": "2025-12-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": "arXiv v1 2025-12-11, v2 2026-06-01; the title is the same in both versions. Venue: v2 arXiv comment and the GitHub description say CVPR 2026 Oral. Models benchmarked (README list, 10): MagicDrive, Panacea, DreamForge, DriveDreamer-2, DrivingSphere, OpenDWM, MagicDrive-V2, DiST-4D, RLGF, X-Scene; Generation, Reconstruction, Downstream and Human tables cover 6 models, Action-Following covers 6 (Panacea only on displacement error). Displacement error uses 150 nuScenes validation scenes; closed-loop tests use two maps (singapore-onenorth, boston-seaport) and five simulation sequences. Human annotation: ten annotators in two independent groups, about 2 min 8 s per annotation, over 930 hours. Leaderboard: Hugging Face Space worldbench/WorldLens (Gradio, running, created 2025-12-09; holds result files for 9 models). GitHub stars: 256 on 2026-10-10. 'Is your driving world model an all-around player?' is the Figure 1 caption and project-page tagline, not a paper title."
   },
   {
    "name": "DrivingGen",
    "kind": "benchmark",
    "domains": [
     "driving"
    ],
    "org": "University of Toronto; CUHK MMLab",
    "date": "2026-01-04",
    "measures": "Generative video world models for driving (image-to-video, optionally conditioned on an ego trajectory): how real the video looks, how plausible the ego motion implied by the video is, temporal and per-agent consistency, and how closely the video follows a commanded ego trajectory, across varied weather, time of day, world regions and maneuvers.",
    "scoring": "Metrics on 100-frame generated videos in four groups. Distribution: FVD, and a new Frechet Trajectory Distance (FTD) computed with a Motion Transformer encoder on trajectories recovered from the video (SIFT/RANSAC PnP with UniDepthV2 depth). Quality: CLIP-IQA+ image quality, the IEEE P2020 Modulation Mitigation Probability for flicker, and a composite trajectory quality score (comfort, motion, curvature). Temporal consistency: motion-adaptive DINOv3 frame consistency, agent appearance consistency (YOLOv10 detection, SAM2 tracking), abnormal agent disappearance judged by the Cosmos-Reason1 vision-language model, and trajectory speed/acceleration stability. Trajectory alignment (ego-conditioned track only): ADE and DTW. All metrics are reported; an average rank is given as a summary.",
    "licence": "Code: Apache-2.0 (LICENSE in github.com/youngzhou1999/DrivingGen, read via gh api 2026-10-11). Data: Hugging Face dataset yangzhou99/DrivingGen card states apache-2.0. The paper says the samples come from internet video and from five driving datasets (ZOD, DrivingDojo, CoVLA, nuPlan, WOMD); the card does not mention those sources' own terms.",
    "url": "https://drivinggen-bench.github.io/",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2601.01528",
      "title": "DrivingGen: A Comprehensive Benchmark for Generative Video World Models in Autonomous Driving (arXiv abs)",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML full text v2 (incl. Appendix B.8, B.9 and Figure 5)",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://drivinggen-bench.github.io/",
      "title": "DrivingGen project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://github.com/youngzhou1999/DrivingGen",
      "title": "youngzhou1999/DrivingGen GitHub repository",
      "type": "repo",
      "date": "2026-03-13",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/datasets/yangzhou99/DrivingGen",
      "title": "DrivingGen dataset card",
      "type": "dataset-card",
      "date": "2026-06-05",
      "accessed": "2026-10-11"
     }
    ],
    "note": "arXiv v1 2026-01-04, v2 2026-03-07. Venue: ICLR 2026 Poster (arXiv comment, repo description, project page; an 'ICCV 2025' string on the project page sits inside an HTML comment left from a template and is not displayed). Paper: 400 samples, 200 per track (open-domain track from internet video; ego-conditioned track from five datasets). Conflict: the Hugging Face card lists 222 examples in each of its two configs (open_domain, ego_condition). 14 models evaluated: Gen-3 and Kling (closed-source), CogVideoX, Wan, HunyuanVideo, LTX-Video, SkyReels, Cosmos-Predict1, Cosmos-Predict2, Vista, DrivingDojo, GEM, VaViM, UniFuture. Leaderboard is marked 'Coming' on the project page. Running all metrics on 400 videos takes about 1 to 2 days on one GPU (Appendix B.8); generation itself is slower (Wan2.2-14B about 20 to 30 minutes per 100-frame video). Open-loop only; the authors list closed-loop evaluation as future work. GitHub stars: 43 on 2026-10-11."
   },
   {
    "name": "GameWorld Score",
    "kind": "benchmark",
    "domains": [
     "games"
    ],
    "org": "Skywork AI",
    "date": "2025-06-23",
    "measures": "Minecraft world models that generate video from an initial image plus keyboard and mouse actions: visual quality, temporal quality, action controllability and physical rule understanding, in 8 dimensions.",
    "scoring": "Image quality (MUSIQ), aesthetic quality (LAION aesthetic predictor), temporal consistency (CLIP similarity of adjacent frames), motion smoothness (frame-interpolation reconstruction error), keyboard and mouse accuracy (an inverse dynamics model infers actions from the generated video, compared with the input actions; precision over four keyboard groups and nine camera-direction classes), object consistency (DROID-SLAM reprojection error, following WorldScore), scenario consistency (MSE between frames of symmetric go-and-return camera motions, allowing 4-pixel shifts). Scores are reported per dimension on a 0 to 1 scale; there is no single aggregate.",
    "licence": "Code: MIT (repo-root LICENSE of github.com/SkyworkAI/Matrix-Game, read via gh api 2026-10-10); the Matrix-Game-1/GameWorldScore folder contains its own Apache License 2.0 file, so the benchmark code carries Apache-2.0 at folder level. Data: the initial test images are in Matrix-Game-1/GameWorldScore/asset/init_image (8 biome folders) in the same repository.",
    "url": "https://arxiv.org/abs/2506.18701",
    "in_atlas": false,
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2506.18701",
      "title": "Matrix-Game: Interactive World Foundation Model (arXiv abs, technical report)",
      "type": "paper",
      "date": "2025-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.18701",
      "title": "Matrix-Game arXiv HTML full text v1 (Section 5, Table 2, Figure 8)",
      "type": "paper",
      "date": "2025-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/SkyworkAI/Matrix-Game/tree/main/Matrix-Game-1/GameWorldScore",
      "title": "GameWorldScore folder in SkyworkAI/Matrix-Game (README, LICENSE, assets)",
      "type": "repo",
      "date": "2026-09-29",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Introduced in the Matrix-Game technical report (arXiv 2506.18701, single version). Test set per the GameWorldScore README: 76 actions x 32 initial images = 2,432 videos for most metrics; scenario consistency uses the same 32 images with 8 mirrored actions. Models evaluated in the paper: Oasis, MineWorld and Matrix-Game (3). Results (Table 2), Oasis / MineWorld / Matrix-Game: image quality 0.65 / 0.69 / 0.72; aesthetic 0.48 / 0.47 / 0.49; temporal consistency 0.94 / 0.95 / 0.97; motion smoothness 0.98 / 0.98 / 0.98; keyboard 0.77 / 0.86 / 0.95; mouse 0.56 / 0.64 / 0.95; object consistency 0.56 / 0.51 / 0.76; scenario consistency 0.86 / 0.92 / 0.93. The paper states the inverse dynamics model was trained on 1,962 hours of Minecraft gameplay and reaches 90.6% keyboard accuracy and R^2 0.97 for mouse movement (numbers cited from earlier work, not re-measured). The benchmark was built by the same team that built Matrix-Game. No leaderboard found. Repo stars: 2,348 on 2026-10-10 (repo now hosts Matrix-Game 1.0, 2.0 and 3.0)."
   }
  ],
  "validity_studies": [
   {
    "name": "PAI-Bench-G scores vs human pairwise preferences",
    "date": "2025-12",
    "world_model": "Video generators scored by PAI-Bench-G (not a single world model)",
    "compared_with": "human ratings",
    "comparison_detail": "Elo ratings from an arena-style pairwise human study (separate quality and physical-plausibility votes) against PAI-Bench-G Quality and Domain scores.",
    "statistic": "Pearson r = 0.918 (overall)",
    "n_policies": "not applicable: video generators (the atlas entry records source videos + 8 models)",
    "by": "authors",
    "atlas_id": "pai-bench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.01989v1",
      "title": "PAI-Bench: A Comprehensive Benchmark For Physical AI",
      "type": "paper",
      "date": "2025-12-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Agreement with human ratings only; no robot outcome was measured."
   },
   {
    "name": "RBench scores vs human pairwise preferences",
    "date": "2026-01",
    "world_model": "Video generators scored by RBench",
    "compared_with": "human ratings",
    "comparison_detail": "30 participants made A/B/tie choices between videos from two models for the same prompt; votes converted to per-model scores (win 5, tie 3, loss 1).",
    "statistic": "Spearman rho = 0.96 (two-sided p < 10^-3) on a ten-model subset; a Bland-Altman analysis is also reported",
    "n_policies": "not applicable: 10 video generators",
    "by": "authors",
    "atlas_id": "rbench",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.15282v1",
      "title": "Rethinking Video Generation Model for the Embodied World (RBench)",
      "type": "paper",
      "date": "2026-01-21",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The NVIDIA Cosmos 3 report restates this as rho = 0.96 'across 25 evaluated models'; the RBench paper measured it on 10 of its 25 models."
   },
   {
    "name": "WorldEval vs real-robot success rates",
    "date": "2025-05-25",
    "world_model": "WorldEval: Wan 2.1 14B image-to-video model fine-tuned with LoRA and conditioned on Policy2Vec latent actions (action-conditioned video model)",
    "compared_with": "real robots",
    "comparison_detail": "Per-policy real-robot success rates of 4 policies within each task, 40 real rollouts per task on an AgileX bimanual arm; world-model success judged by Gemini-2.0. Table 1 covers 3 tasks; Figure 4 covers all 5 tasks.",
    "statistic": "Table 1 (3 tasks): Pearson r = 0.958 (Place Cup), 0.887 (Strike Block), 0.980 (Handover Block), average 0.942; MMRV = 0.000, 0.133, 0.000, average 0.044. Figure 4 (5 tasks): Bussing Table r = 0.935, MMRV = 0.000; Collect Toy r = 0.885, MMRV = 0.133; Place Cup r = 0.958, MMRV = 0.000; Handover Block r = 0.980, MMRV = 0.000; Strike Block r = 0.887, MMRV = 0.000.",
    "n_policies": 4,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Table 1, Section 4",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017v1/muti_task_corr.png",
      "title": "WorldEval Figure 4: real vs WorldEval success rates per task",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Conflict inside the paper: Table 1 gives Strike Block MMRV 0.133, while Figure 4 gives Strike Block MMRV 0.000 and Collect Toy MMRV 0.133. Each per-task r is computed over 4 policies. Real-robot trials were newly run by the authors (Midea Group). The paper does not say whether the 40 rollouts are per policy. Collect Toy uses objects and an instruction absent from training. The paper reports no direction of bias. The real success rates plotted for the same policies differ between Figure 4 and Figure 7 (for example pi0 on Bussing Table is at about 1.0 in Figure 4 and about 0.8 in Figure 7; read from the figures, inferred). The abstract claims checkpoint ranking, but Appendix Table 3 gives success rates and FID for checkpoints without a real-robot agreement statistic, and it does not say whether its success rates are real or generated.",
    "r_main": 0.958
   },
   {
    "name": "RoboTwin real-to-sim baseline vs real-robot success rates (in the WorldEval paper)",
    "date": "2025-05-25",
    "world_model": "none: RoboTwin physics simulator with SIMPLER-style visual matching through MidJourney image translation (comparison point)",
    "compared_with": "real robots",
    "comparison_detail": "Policies trained on real data were evaluated in RoboTwin on the 3 tasks adapted from RoboTwin (Place Cup, Strike Block, Handover Block); simulator success rates compared with the same real-robot success rates used for WorldEval.",
    "statistic": "Pearson r = 0.328 (Place Cup), 0.442 (Strike Block), 0.465 (Handover Block), average 0.411; MMRV = 0.332, 0.266, 0.187, average 0.261",
    "n_policies": "4 (implied; the paper does not restate the policy list for this comparison)",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Table 1, Section 4.2, Appendix C",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2504.13059",
      "title": "RoboTwin (arXiv abs, author list)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Independent of RoboTwin's builders (no author overlap between the RoboTwin and WorldEval author lists), but measured by the WorldEval authors, who propose the competing method. Simulation images were passed through MidJourney to look more realistic before reaching the policy.",
    "r_main": 0.328
   },
   {
    "name": "WorldEval action-encoding ablation vs real-robot results",
    "date": "2025-05-25",
    "world_model": "WorldEval video model with three action encodings: Policy2Vec, VQ-VAE, one-hot",
    "compared_with": "real robots",
    "comparison_detail": "Agreement of each world-model variant with real-robot success rates across five tasks (aggregation method not stated).",
    "statistic": "Policy2Vec: Pearson r = 0.939, MMRV = 0.192, FID = 61.33; VQ-VAE: r = -0.862, MMRV = 0.292, FID = 71.79; one-hot: r = -0.333, MMRV = 0.416, FID = 75.91 (Table 2)",
    "n_policies": "4 (implied)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Table 2, Section 4.3",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper does not explain why Policy2Vec's five-task MMRV here (0.192) is higher than the per-task values in Table 1 and Figure 4 (0.000 to 0.133). The text says Policy2Vec reduces MMRV by '0.1 and 2.24' versus VQ-VAE and one-hot; the table values give differences of 0.100 and 0.224.",
    "r_main": 0.939
   },
   {
    "name": "WorldEval in unseen backgrounds vs real-robot results",
    "date": "2025-05-25",
    "world_model": "WorldEval (same model; policies trained only on lab data)",
    "compared_with": "real robots",
    "comparison_detail": "Real-robot and WorldEval success rates for Place Cup, Strike Block and Handover Block in three new settings (office, living room, kitchen).",
    "statistic": "MMRV = 0.047; Pearson r = 0.927",
    "n_policies": "not stated (4 in the main study)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Appendix B",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper compares this with the in-lab MMRV of 0.044. Number of real rollouts in the new settings is not stated.",
    "r_main": 0.927
   },
   {
    "name": "FID of WorldEval videos vs real-robot success rates",
    "date": "2025-05-25",
    "world_model": "WorldEval video model, scored by FID of generated videos instead of a success judge",
    "compared_with": "real robots",
    "comparison_detail": "Correlation between FID of each policy's generated videos and its real-robot success rate, per task, 4 policies.",
    "statistic": "|r| = 0.495 (Bussing Table), 0.760 (Collect Toy), 0.999 (Place Cup), 0.993 (Handover Block), 0.805 (Strike Block) (Figure 7)",
    "n_policies": 4,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017v1/real_sucess_fid_corr.png",
      "title": "WorldEval Figure 7: real success rate vs FID",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Section 4.3",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The authors recommend FID only for simple tasks with one or two objects and note the lower correlation on Bussing Table.",
    "r_main": null
   },
   {
    "name": "WorldGym vs OpenVLA Bridge real-robot results",
    "date": "2025-09-30",
    "world_model": "WorldGym: autoregressive latent diffusion transformer trained with Diffusion Forcing on Open X-Embodiment robot data (action-conditioned video model)",
    "compared_with": "real robots",
    "comparison_detail": "Real-world success rates of RT-1-X, Octo and OpenVLA on the 17-task OpenVLA Bridge (WidowX) suite, 10 trials per task per policy, copied from the OpenVLA paper; WorldGym reruns each trial from its recorded first frame and GPT-4o scores the rollout.",
    "statistic": "Pearson r = 0.78 between per-task success rates in WorldGym and in the real world (each point is one task-policy pair). Mean success rate, real vs WorldGym: RT-1-X 18.5% vs 15.5%, Octo 20.0% vs 23.82%, OpenVLA 70.6% vs 67.4%; average difference 3.3%. The order of the three policies by mean success rate is the same in both.",
    "n_policies": 3,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym full text (arXiv HTML v3): Section 4.1, Table 5, Appendix B.2",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-model-eval.github.io/abstract.html",
      "title": "WorldGym project page (r = 0.78, 3.3% mean difference)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2406.09246",
      "title": "OpenVLA (arXiv abs, author list)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "No new real-robot trials: Table 5 states the real success rates are taken directly from Kim et al. (OpenVLA). Percy Liang is an author of both WorldGym and OpenVLA. Recomputed from Table 5 over the 51 task-policy pairs, Pearson r = 0.785 (inferred). Recomputed within each policy across its 17 tasks, r = 0.52 (RT-1-X), 0.47 (Octo), 0.44 (OpenVLA) (inferred), so per-task agreement within one policy is lower than the pooled figure. WorldGym is below the real mean for RT-1-X and OpenVLA and above it for Octo, so no single bias direction. Ordering checks with no statistic: Octo-Small 1.5 vs Octo-Base 1.5 and OpenVLA v0.1 7B vs OpenVLA 7B agree with orderings reported in earlier papers; checkpoint sweeps (video policy at 2K-18K steps, DexVLA-recipe diffusion policy at 10K-60K steps) were compared with validation MSE, not with real robots. Google Robot rollouts (Table 4) have no real-world comparison. GPT-4o judge checked on real RT-1 videos: true positive rate 0.81 +/- 0.14, false positive rate 0.03 +/- 0.05.",
    "r_main": 0.78
   },
   {
    "name": "Ctrl-World vs real-robot rollouts on the authors' DROID setup",
    "date": "2025-10-11",
    "world_model": "Ctrl-World: multi-view action-conditioned video world model fine-tuned from Stable Video Diffusion on DROID",
    "compared_with": "real robots",
    "comparison_detail": "Instruction-following rate and task success rate of 3 policies (pi0-droid, pi0-FAST-droid, pi0.5-droid) on 7 tasks on the authors' own DROID platform (Franka Panda); real and world-model rollouts start from the same initial observations; human annotators label both.",
    "statistic": "No correlation coefficient or MMRV reported. Linear fits of world-model rate on real rate across policy-task pairs (Figure 7): instruction following y = 0.87x - 0.04; success rate y = 0.81x - 0.11.",
    "n_policies": 3,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2510.10125v3",
      "title": "Ctrl-World PDF v3: Section 5.3, Figure 7",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.10125",
      "title": "Ctrl-World full text (arXiv HTML v3): Appendix B Table 3",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Robert-gyj/Ctrl-World",
      "title": "Ctrl-World repository readme (20 runs per task category)",
      "type": "repo",
      "date": "2025-10-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-robot rollouts were newly run by the authors. Figure 7 caption: the world model 'tends to underestimate the execution success rate'. From Table 3 (inferred): world-model success rate is below the real rate in all 21 policy-task pairs (mean 0.324 vs 0.526); Pearson r recomputed over the 21 pairs is 0.97 for instruction following and 0.83 for success rate; the policy order pi0 < pi0-FAST < pi0.5 is the same in both settings for both measures. Rollouts per pair are not stated in the paper; the repository readme says each task category was run 20 times. The authors name gaps in collisions, objects sliding away, rotations, and policies retrying after failure.",
    "r_main": null
   },
   {
    "name": "Ctrl-World as a baseline evaluator in the PolaRiS study",
    "date": "2025-12-18",
    "world_model": "Ctrl-World (open-source action-conditioned video world model, used as released)",
    "compared_with": "real robots",
    "comparison_detail": "PolaRiS paired real-world evaluations of DROID policies in 6 environments at UW and Princeton (20 real rollouts per policy per environment, 0-1 progress rubric graded by a human), compared with human-scored rollouts of the same policies in Ctrl-World.",
    "statistic": "Pearson r = 0.53; MMRV = 0.22 (PolaRiS Figure 7). Same figure, other evaluators: PolaRiS r = 0.90, MMRV = 0.03; LIBERO-90 fine-tuned checkpoints at 1k / 10k / 50k steps r = 0.66 / 0.70 / 0.66, MMRV = 0.19 / 0.04 / 0.15; action MSE r = -0.55 (train) and -0.53 (validation), MMRV = 0.40 for both.",
    "n_policies": "4 policies listed in PolaRiS Section 5.1 (pi0, pi0-FAST, PaliGemma-binning, pi0.5); the policy set and number of Ctrl-World rollouts used for this baseline are not stated",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS full text (arXiv HTML v2): Sections 5.1-5.2",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.16881v2/figures/main_barplots.png",
      "title": "PolaRiS Figure 7: Pearson r and MMRV by evaluation method",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World (arXiv abs, author list)",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Not independent: Chelsea Finn is an author of both PolaRiS and Ctrl-World. Real-robot numbers were newly collected by the PolaRiS authors. PolaRiS reports that Ctrl-World 'often produces heavy hallucinations during object interaction', which made scoring hard and led to policy mis-rankings. The paper does not say whether Ctrl-World was adapted to the test scenes.",
    "r_main": 0.53
   },
   {
    "name": "PolaRiS simulator vs paired real-robot evaluations",
    "date": "2025-12-18",
    "world_model": "none: Gaussian-splat simulator (comparison point)",
    "compared_with": "real robots",
    "comparison_detail": "Per-environment progress scores of DROID policies in 6 simulated replicas vs the 6 real environments (UW and Princeton); 20 real and 50 simulated rollouts per policy-task pair.",
    "statistic": "Average Pearson r = 0.90 and MMRV = 0.03 (Figure 7). Per environment (Figure 13): Block Stacking r = 0.84, MMRV = 0.07; Food Bussing r = 0.81, MMRV = 0.07; Move Latte Cup r = 0.89, MMRV = 0.00; Organize Tools r = 0.93, MMRV = 0.01; Pan Cleaning r = 0.99, MMRV = 0.03; Tape Into Container r = 0.94, MMRV = 0.00.",
    "n_policies": "4 policies (pi0, pi0-FAST, PaliGemma-binning, pi0.5); Figure 6 also plots a pi0 100k-step variant",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS full text (arXiv HTML v2): Sections 1, 5.1-5.3",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.16881v2/figures/correlation_per_env.png",
      "title": "PolaRiS Figure 13: per-environment Pearson r and MMRV",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.16881v2/figures/main-correlation.png",
      "title": "PolaRiS Figure 6: real vs sim scatter with policy list",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Comparison point, not a learned world model. Real-robot rollouts were newly run by the authors and graded by a human. Policies are co-fine-tuned for 1k steps on simulated data before simulated evaluation. The authors state the simulated tasks cover rigid-object manipulation and leave out soft-body and complex contact behaviour.",
    "r_main": 0.9
   },
   {
    "name": "PolaRiS simulator vs RoboArena real-world scores",
    "date": "2025-12-18",
    "world_model": "none: Gaussian-splat simulator (comparison point)",
    "compared_with": "real robots",
    "comparison_detail": "PolaRiS scores vs average progress scores of the same policies in the RoboArena distributed real-world evaluation.",
    "statistic": "Pearson r = 0.98; MMRV = 0.00 (Figure 8)",
    "n_policies": 4,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881v2/figures/roboarena_correlation.png",
      "title": "PolaRiS Figure 8: PolaRiS vs RoboArena",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2506.18123",
      "title": "RoboArena (arXiv abs, author list)",
      "type": "paper",
      "date": "2025-06-22",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Comparison point, not a learned world model. RoboArena numbers were taken from RoboArena, not newly collected. RoboArena and PolaRiS share authors (Arhan Jain, Karl Pertsch, Kanav Arora, Marcel Torne, Abhishek Gupta, Sergey Levine, Chelsea Finn). The authors note the simulated tasks cover only a small part of what RoboArena tests.",
    "r_main": 0.98
   },
   {
    "name": "Veo (Robotics) nominal scenes vs real ALOHA 2 evaluations",
    "date": "2025-12-11",
    "world_model": "Veo (Robotics): Veo 2 video model fine-tuned for robot-pose conditioning and four-view generation (action-conditioned video model)",
    "compared_with": "real robots",
    "comparison_detail": "Per-checkpoint success rates over 80 scene-instruction combinations from 5 ALOHA 2 tasks; binary success; human scoring of 8-second generated rollouts.",
    "statistic": "Pearson = 0.88; MMRV = 0.03 (Figure 4)",
    "n_policies": "8 checkpoints of Gemini Robotics On-Device (GROD) policies",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 3",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/nominal_correlation.png",
      "title": "Veo report Figure 4: nominal real vs predicted success rates",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-robot evaluations were newly run by the same Google DeepMind team (1600+ real evaluations across all experiments). Predicted absolute success rates are lower than real ones: in Figure 4 predicted rates span about 0.01-0.30 and real rates about 0.05-0.69 (read from the figure).",
    "r_main": 0.88
   },
   {
    "name": "Veo (Robotics) ranking of generalization conditions for one policy",
    "date": "2025-12-11",
    "world_model": "Veo (Robotics), with scenes edited by Gemini 2.5 Flash Image and completed to four views by a multi-view Veo 2 model",
    "compared_with": "real robots",
    "comparison_detail": "Success rates of one checkpoint (Policy A) under 5 conditions: nominal, new background, small distractor, large distractor, novel object to manipulate; edited scenes were recreated physically for real trials.",
    "statistic": "Pearson = 0.86; MMRV = 0.06 (Figure 8)",
    "n_policies": "1 checkpoint (5 conditions)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 4.1",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/ood_policy_A.png",
      "title": "Veo report Figure 8: Policy A across generalization axes",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The unit of comparison is the condition, not the policy. Both numbers come from this single 5-point comparison. Predicted success rates are lower than real ones. The model ranks the novel-object condition as the largest drop, which the real trials match.",
    "r_main": 0.86
   },
   {
    "name": "Veo (Robotics) policy comparison within each out-of-distribution axis",
    "date": "2025-12-11",
    "world_model": "Veo (Robotics), with edited scenes as above",
    "compared_with": "real robots",
    "comparison_detail": "Per-checkpoint success rates for each axis (background, small distractor, large distractor, novel object), real trials in physically recreated scenes.",
    "statistic": "Background: Pearson = 0.91, MMRV = 0.0; Small distractor: Pearson = 0.86, MMRV = 0.10; Large distractor: Pearson = 0.77, MMRV = 0.14; Object: Pearson = 0.56, MMRV = 0.15 (Figure 9)",
    "n_policies": "5 checkpoints",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 4.2",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/ood_policy_comparison.png",
      "title": "Veo report Figure 9: policy comparison per OOD axis",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The authors note that all policies have low success with novel objects, which makes them harder to tell apart. Each r rests on 5 points.",
    "r_main": 0.91
   },
   {
    "name": "Veo (Robotics) safety red-teaming vs real props",
    "date": "2025-12-11",
    "world_model": "Veo (Robotics), with generated hazard scenes",
    "compared_with": "real robots",
    "comparison_detail": "Scenes with hazards and ambiguous requests, generated and filtered with Gemini 2.5 Pro as critic; unsafe behaviours predicted for Policy A were recreated with real props.",
    "statistic": "No statistic reported.",
    "n_policies": "1 checkpoint (Policy A)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 5",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://veo-robotics.github.io",
      "title": "Veo Robotics project page (safety examples)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper shows 2 examples (gripper contacting a human hand on 'Quick, grab the red block!'; closing a laptop without first moving scissors out of the way); the project page adds a third ('I'm thirsty, grab the bottle'). The authors report the predicted unsafe behaviours were observed in the real replications. The number of scenarios generated or replicated is not stated.",
    "r_main": null
   },
   {
    "name": "GWM-Robotics vs RoboArena real-world results",
    "date": "2026-02-27",
    "world_model": "GWM-Robotics: Runway world model, a variant of the Gen-4.5 video generation model, fine-tuned on robot data (action-conditioned video model)",
    "compared_with": "real robots",
    "comparison_detail": "Per-policy progress scores from 1,450 simulated rollouts (initial conditions from earlier RoboArena evaluations, Franka Panda), graded by humans (over 16,000 ratings, about 10 graders per rollout), vs the same policies' real-world RoboArena rollouts.",
    "statistic": "Pearson correlation = 0.95 (figure label: Pearson r = 0.952); MMRV = 0.033",
    "n_policies": 8,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://d3phaj0sisr2ct.cloudfront.net/research/images/real_vs_gwm1_success_rates-01.png",
      "title": "Runway figure: Real vs. GWM-1 Success Rates",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-world numbers come from the RoboArena dataset; Runway ran no new real-robot trials. Policies in the figure: Pi-0.5, Pi-0 Fast, Pi-0, PaLI-Gemma Fast, PaLI-Gemma Specialist, PaLI-Gemma Diffusion, PaLI-Gemma VQ, PaLI-Gemma Binning. The text describes progress scores while the figure axes say success rate. In the figure the simulated rate is above the real rate for 7 of 8 policies and about equal for PaLI-Gemma Fast; for PaLI-Gemma Binning it is about 19 vs about 6 (read from the figure, inferred). The post says human graders identified the best (pi05_droid) and worst (paligemma_binning_droid) policies. No paper, code or data release found.",
    "r_main": 0.95
   },
   {
    "name": "GWM-Robotics vs RoboArena on the policies PolaRiS also evaluated",
    "date": "2026-02-27",
    "world_model": "GWM-Robotics (as above)",
    "compared_with": "real robots",
    "comparison_detail": "Same comparison as above, restricted to the RoboArena policy architectures that PolaRiS also evaluated; Runway sets this against PolaRiS's own r = 0.98 versus RoboArena.",
    "statistic": "Pearson correlation = 0.986; MMRV = 0",
    "n_policies": "not stated (PolaRiS's RoboArena comparison used 4 policies)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The real-world reference is RoboArena for both GWM-Robotics and PolaRiS. The PolaRiS figure of 0.98 comes from the PolaRiS paper, with different rollouts and automatic scoring, so the two numbers were not produced under one protocol (inferred).",
    "r_main": 0.986
   },
   {
    "name": "Cosmos-Surg-dVRK (human labels) vs real dVRK success rates",
    "date": "2025-10-17",
    "world_model": "Cosmos-Surg-dVRK: Cosmos-Predict2-2B-Video2World fine-tuned on dVRK video and kinematics (action-conditioned video world foundation model)",
    "compared_with": "real robots",
    "comparison_detail": "Success rates of 6 checkpoints on 4 tabletop suture-pad tasks; 10 real rollouts per task per checkpoint on dVRK Si; 10 world-model trials per task, each generated with 3 seeds; two human raters label world-model videos and their scores are averaged.",
    "statistic": "Pooled Pearson r = 0.718 (p < 0.001) across all tasks and training regimes. Per task Pearson / MMRV: Handover 0.468 / 0.217; Throw 0.716 / 0.183; Knot Tie 0.840 / 0.050; Pickup 0.806 / 0.067; average 0.707 / 0.129 (Table 2: Pearson 0.71 +/- 0.17, MMRV 0.13 +/- 0.08). Mean bias error 0.140 (95% CI 0.081-0.199).",
    "n_policies": "6 checkpoints from 3 policies (pi0, GR00T N1, GR00T N1.5, each at half and full training)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Sections 4.1-5.2.1, Tables 1-2",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-robot trials were newly run by the authors. Positive mean bias error means the world model gives higher success rates than the real robot. The two human raters agree with ICC(2,1) = 0.811. The pooled r appears to use the 24 task-checkpoint points shown in Figure 4 (inferred).",
    "r_main": 0.718
   },
   {
    "name": "Cosmos-Surg-dVRK (V-JEPA 2 classifier) vs real dVRK success rates",
    "date": "2025-10-17",
    "world_model": "Cosmos-Surg-dVRK (as above), with success labelled by a V-JEPA 2 attentive-probe classifier",
    "compared_with": "real robots",
    "comparison_detail": "Same rollouts and real success rates as the human-label comparison; classifier labels averaged over 3 seeds.",
    "statistic": "Pooled Pearson r = 0.756 (p < 0.001). Per task Pearson / MMRV: Handover 0.656 / 0.133; Throw 0.639 / 0.117; Knot Tie 0.729 / 0.033; Pickup 0.639 / 0.100; average 0.666 / 0.096 (text and Table 2: MMRV 0.10 +/- 0.04, Pearson 0.67 +/- 0.04). Mean bias error 0.153 (95% CI 0.093-0.213).",
    "n_policies": "6 checkpoints from 3 policies",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Section 5.2.2, Tables 1-2",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Classifier: attentive probe on a frozen V-JEPA 2 ViT-H, trained on 2,310 manually labelled clips. Classifier vs human labels on the same videos: ICC(2,1) = 0.836, Pearson r = 0.840 (p < 0.001).",
    "r_main": 0.756
   },
   {
    "name": "Cosmos-Surg-dVRK trained without failure episodes vs real dVRK",
    "date": "2025-10-17",
    "world_model": "Cosmos-Surg-dVRK variant fine-tuned on successful episodes only",
    "compared_with": "real robots",
    "comparison_detail": "Hold-out evaluation, one seed, 10 rollouts per task and checkpoint, human labels averaged over two raters.",
    "statistic": "Per task Pearson / MMRV: Handover 0.313 / 0.183; Throw 0.533 / 0.317; Knot Tie 0.922 / 0.017; Pickup 0.701 / 0.067; average 0.617 / 0.146 (Table 2: Pearson 0.62 +/- 0.26, MMRV 0.15 +/- 0.13). Mean bias error 0.325 (95% CI 0.262-0.388).",
    "n_policies": "6 checkpoints from 3 policies",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Section 5.2.4, Tables 1-2",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The authors conclude that failure episodes in the fine-tuning data reduce the positive success bias (MBE 0.140 with failures vs 0.325 without).",
    "r_main": null
   },
   {
    "name": "Cosmos-Surg-dVRK vs real dVRK on ex-vivo porcine cholecystectomy",
    "date": "2025-10-17",
    "world_model": "Cosmos-Surg-dVRK fine-tuned on the cholecystectomy dataset",
    "compared_with": "real robots",
    "comparison_detail": "SRT-H policy (monocular, 'no wrist camera' ablation) on 3 tasks with 3 trials each from a hold-out set of nine porcine tissues; world-model outcome per trial by majority over 3 seeds x 2 raters.",
    "statistic": "No correlation statistic. Success counts, real dVRK vs Cosmos-Surg-dVRK: apply first clip 3/3 vs 3/3; apply third clip 2/3 vs 3/3; cut cystic duct 2/3 vs 2/3; total 7/9 vs 8/9.",
    "n_policies": 1,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Sections 4.1.2 and 5.3, Table 3",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-robot results were copied from Kim et al. (2025), not newly run; Ji Woong Kim is an author of both papers. The authors call these results exploratory.",
    "r_main": null
   },
   {
    "name": "1X World Model outcome-prediction alignment",
    "date": "2025-06-16",
    "world_model": "1X World Model (1XWM): generative video world model with a state-value head, conditioned on humanoid actions",
    "compared_with": "real robots",
    "comparison_detail": "Accuracy of success/failure predictions against real outcomes on held-out episodes of the Shelf task (EVE robot), with and without added Arcade task data.",
    "statistic": "Alignment 63.06% when trained on about 216M Shelf video tokens; 71.17% with an added about 1.46B Arcade video tokens",
    "n_policies": "not a per-policy comparison (per-episode outcome prediction)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Sections 5.1-5.3",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "50% equals chance. Figure 5 also shows alignment rising with more Airfryer (NEO) and Arcade (EVE) data, without exact values in the text.",
    "r_main": null
   },
   {
    "name": "1X World Model checkpoint and architecture selection vs double-blind real A/B evaluations",
    "date": "2025-06-16",
    "world_model": "1X World Model (1XWM)",
    "compared_with": "real robots",
    "comparison_detail": "1XWM scores for checkpoints on the Arcade task, shown in plots next to real evaluations run as double-blind A/B experiments on one robot in one setting.",
    "statistic": "No correlation statistic reported. Analytic claim: with a true real success-rate gap of 15% between two checkpoints, a world model with 70% alignment picks the better one with 90% success.",
    "n_policies": "checkpoint sweeps of 2 policy architectures (Figure 7); ViT-B vs ViT-L and with vs without proprioception (Figure 8); 4 base architectures x 2 sampling strategies (Figure 9)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Section 6",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.1x.tech/discover/redwood-ai-world-model",
      "title": "1X World Model (1X blog)",
      "type": "blog",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The report states that a checkpoint scoring clearly higher in the world model tends to score higher in real evaluation, with different margins. The blog words the analytic claim with 'policies' and '70% accuracy'. Real evaluations were run by 1X.",
    "r_main": null
   },
   {
    "name": "RoboWorld vs RoboArena leaderboard",
    "date": "2026-07-01",
    "world_model": "RoboWorld: autoregressive video world model adapted from Wan2.1-T2V-1.3B with Step Forcing, trained on DROID",
    "compared_with": "real robots",
    "comparison_detail": "RoboWorld scores of 8 open-source DROID policies from 4,186 rollouts started from RoboArena episode initial frames (data dump 2026-02-03), vs the RoboArena real-world leaderboard (snapshot 2026-02-26); GPT-4o with a 0-5 progress rubric.",
    "statistic": "Pearson r = 0.989; Spearman rho = 0.970",
    "n_policies": 8,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Sections 1, 5.3",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2506.18123",
      "title": "RoboArena (arXiv abs, author list)",
      "type": "paper",
      "date": "2025-06-22",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Real-world numbers are the RoboArena leaderboard; no new real-robot trials. No author overlap found between RoboWorld and the RoboArena paper. Replicating the benchmark for 8 policies took 100 H100 GPU hours.",
    "r_main": 0.989
   },
   {
    "name": "RoboWorld scoring and judge variants vs RoboArena leaderboard",
    "date": "2026-07-01",
    "world_model": "RoboWorld (as above)",
    "compared_with": "real robots",
    "comparison_detail": "Same rollouts and leaderboard; scoring rubric, camera view used for success, and judge model varied.",
    "statistic": "Binary success scoring: Spearman rho = 0.922; wrist view used for success judgments: rho = 0.862; Gemini-2.5-Flash as judge: Pearson r = 0.944 (p < 0.001); RoboWorld success-rate metric: r = 0.901 (p = 0.002, GPT-4o) and r = 0.876 (p = 0.004, Gemini-2.5-Flash)",
    "n_policies": 8,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Section 5.4, Appendix B.4-B.5",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The authors attribute the wrist-view drop to more generation artifacts in that view.",
    "r_main": 0.944
   },
   {
    "name": "RoboWorld in image-edited environments vs RoboArena leaderboard",
    "date": "2026-07-01",
    "world_model": "RoboWorld (as above)",
    "compared_with": "real robots",
    "comparison_detail": "746 valid initial conditions in 8 image-edited environments (airplane cabin, spacecraft interior, operating room, nuclear facility, underwater station, disaster site, construction site, mine tunnel) made from 175 RoboArena initial observations; compared with the leaderboard from real scenes.",
    "statistic": "Pearson r = 0.970",
    "n_policies": 8,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Section 5.4",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     }
    ],
    "note": "No real-robot trials exist in these environments; the reference is the leaderboard from ordinary real scenes.",
    "r_main": 0.97
   },
   {
    "name": "DreamDojo vs real-robot fruit-packing success",
    "date": "2026-02-06",
    "world_model": "DreamDojo-2B post-trained on AgiBot data (action-conditioned video world model)",
    "compared_with": "real robots",
    "comparison_detail": "Real-robot success rates of policy checkpoints (single-view, state-free GR00T N1.5 variant) on AgiBot fruit packing; 20 scenes; one real rollout of about 80 seconds per scene per checkpoint, simulated in DreamDojo from the same initial frame; success = fruits placed in the bag out of 5; generated rollouts scored by human evaluators.",
    "statistic": "Pearson r = 0.995; MMRV = 0.003",
    "n_policies": "6 checkpoints of one policy (points labelled A-F in Fig. 5a; the text does not state the number)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.06949",
      "title": "DreamDojo full text (arXiv HTML v1), Sec. 4.7 'Downstream Applications' and Limitations",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949v1/correlation_compressed.svg",
      "title": "DreamDojo Fig. 5(a): real vs DreamDojo success rates (points A-F)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real trials run by the authors on an AgiBot robot (model not named). Bias: in Fig. 5a DreamDojo success rates span about 0.06-0.81 while real rates span 0.0-about 0.44 (read from the figure); the paper says 'absolute success rates in DreamDojo are often higher than their real counterparts'. One task only.",
    "r_main": 0.995
   },
   {
    "name": "PlayWorld vs real-robot success (18 policies)",
    "date": "2026-03-09",
    "world_model": "PlayWorld (SVD-based action-conditioned video model fine-tuned on 30 h of autonomous robot play); same architecture trained on human demonstration data and on human play data as baselines",
    "compared_with": "real robots",
    "comparison_detail": "Real-robot success rates of 18 policies (diffusion policies trained from scratch and pi0 fine-tunes with demonstrations of varying quantity and quality) across 3 tasks on a DROID manipulation setup; 20 real trials and 50 world-model trials per policy.",
    "statistic": "PlayWorld: Pearson r = 0.8766, RMSE = 0.171. Human-demo-trained model: r = 0.6636, RMSE = 0.275. Human-play-trained model: r = 0.6619, RMSE = 0.297 (Fig. 7).",
    "n_policies": 18,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.09030",
      "title": "PlayWorld full text (arXiv HTML v3), Sec. 4.4 and Fig. 7",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v3/images/experiment/correlation_new.png",
      "title": "PlayWorld Fig. 7: policy evaluation success-rate correlation (RMSE and r per training-data source)",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v1",
      "title": "PlayWorld arXiv HTML v1 (checked that Pearson 0.8766 and 18 policies already appear)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2603.09030",
      "title": "PlayWorld: Learning Robot World Models from Autonomous Play (arXiv abstract; v1 2026-03-09, v3 2026-04-06)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real trials run by the authors at Princeton. The text states only the PlayWorld Pearson value; RMSE and baseline values come from Fig. 7. The paper reports 'hallucinated success' as the most common failure of the baseline models. The abstract's 'up to 40% improvements over human-collected data' is not tied to a specific number in Sec. 4.4. Arm model not named in the sections read.",
    "r_main": 0.8766
   },
   {
    "name": "PersistWorld vs real-robot task progress",
    "date": "2026-03-26",
    "world_model": "PersistWorld (Ctrl-World post-trained with reinforcement learning on its own autoregressive rollouts; multi-view action-conditioned video diffusion)",
    "compared_with": "real robots",
    "comparison_detail": "Real task progress of 3 policies (pi0, pi0-FAST, GR00T N1.5) on 3 tasks (Put Banana in Box, Put Green Block in Bowl, Rotate Marker): 9 task-policy pairs; 5 real and 11 world-model rollouts per pair; partial-progress rubric.",
    "statistic": "Pearson r = 0.822 (p = 0.007); MMRV = 0.006",
    "n_policies": "3 policies x 3 tasks (9 points)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v1",
      "title": "PersistWorld arXiv HTML v1, Appendix 0.C (same Fig. 8 file as v2)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2603.25685",
      "title": "Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning (arXiv abstract; v1 2026-03-26, v2 2026-09-04; ECCV 2026)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real rollouts collected by the authors; robot not named in the appendix (the world model is trained on DROID, Franka Emika Panda). Bias: both world models make tasks look easier; world-model progress spans about 0.79-1.0 while real progress spans about 0.33-1.0 (read from Fig. 8). Present since v1.",
    "r_main": 0.822
   },
   {
    "name": "Ctrl-World vs real-robot task progress (measured in the PersistWorld paper)",
    "date": "2026-03-26",
    "world_model": "Ctrl-World (pre-trained on DROID; multi-view action-conditioned video diffusion), used as baseline",
    "compared_with": "real robots",
    "comparison_detail": "Same 9 task-policy pairs, rollouts and rubric as the PersistWorld row.",
    "statistic": "Pearson r = 0.796 (p = 0.010); MMRV = 0.053",
    "n_policies": "3 policies x 3 tasks (9 points)",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Measured by Bardhan, Drozdik, Sivic, Petrik (CTU Prague); Ctrl-World authors are Yanjiang Guo, Lucy Xiaoyang Shi, Jianyu Chen, Chelsea Finn, so there is no author overlap. Overestimates progress like PersistWorld.",
    "r_main": 0.796
   },
   {
    "name": "WEAVER vs real-robot success (five tasks)",
    "date": "2026-06-11",
    "world_model": "WEAVER-FT (multi-view latent world model with reward head, pre-trained on DROID and fine-tuned on 50 pi0.5 rollouts per task); also pretrained WEAVER without task fine-tuning",
    "compared_with": "real robots",
    "comparison_detail": "Real success rates of base pi0.5 and a fine-tuned pi0.5 on 5 tasks (Stack Bowls, PnP Bag, PnP Marker, PnP Towel, Pour Beans) on a DROID setup with one Franka Emika Panda; 20 held-out real trials per task; imagined rollouts replay real action sequences open-loop; success on imagined rollouts labelled by humans.",
    "statistic": "WEAVER-FT (Table 8): Pearson = 0.863, Spearman = 0.870, MMRV = 0.035, RMSE = 0.188; abstract and Fig. 6 report rho = 0.870. Pretrained WEAVER: Pearson = 0.563, Spearman = 0.594, MMRV = 0.155, RMSE = 0.359.",
    "n_policies": "2 policies x 5 tasks (10 points)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v2/policy_eval.png",
      "title": "WEAVER Fig. 6: policy evaluation scatter plots (rho and MMRV per world model)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2606.13672",
      "title": "WEAVER, Better, Faster, Longer: An Effective World Model for Robotic Manipulation (arXiv abstract; v1 2026-06-11, v2 2026-06-16)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v1",
      "title": "WEAVER arXiv HTML v1 (checked that rho=0.870 and Table 8 values already appear)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Label conflict: the abstract calls rho = 0.870 the correlation with real-world success rate and Sec. 5.2.1 calls it 'Pearson correlations ... rho=0.87', but Table 8 lists Pearson 0.863 and Spearman 0.870, and Fig. 6's rho values (0.523, 0.594, 0.870) equal the Table 8 Spearman column. Table 8's caption calls the results 'reward prediction quality'. The paper says pretrained world models tend to underestimate policy performance. WEAVER-FT was fine-tuned on rollouts from the same five tasks (separate from the 20 validation rollouts). Real trials by the authors.",
    "r_main": 0.863
   },
   {
    "name": "Ctrl-World vs real-robot success (measured in the WEAVER paper)",
    "date": "2026-06-11",
    "world_model": "Ctrl-World pretrained on DROID (multi-view action-conditioned video diffusion), used as baseline",
    "compared_with": "real robots",
    "comparison_detail": "Same 10 task-policy points and real trials as the WEAVER row.",
    "statistic": "Pearson = 0.552; Spearman = 0.523; MMRV = 0.215; RMSE = 0.410 (Table 8; Fig. 6 shows rho = 0.523, MMRV = 0.215)",
    "n_policies": "2 policies x 5 tasks (10 points)",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v2/policy_eval.png",
      "title": "WEAVER Fig. 6: policy evaluation scatter plots (rho and MMRV per world model)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Measured by Jain, Wu, Farebrother, Swamy, Bajcsy (Mila / CMU / McGill); no author overlap with Ctrl-World (Guo, Shi, Chen, Finn). Underestimates success on most points in Fig. 6.",
    "r_main": 0.552
   },
   {
    "name": "Pelican-Sim 1.0 vs RoboTwin simulator success (five checkpoints)",
    "date": "2026-09-10",
    "world_model": "Pelican-Sim 1.0 (action-conditioned video DiT with sparse MoE, four-step distilled) plus fine-tuned Qwen3-VL-2B-Instruct success judge",
    "compared_with": "other",
    "comparison_detail": "RoboTwin simulator task-checker success rates of 5 checkpoints from one VLA training run; 200 held-out initial conditions per checkpoint; the same policy-predicted action trajectory is run in RoboTwin and supplied to the world model (matched-action, not closed-loop); five world-model adaptation budgets.",
    "statistic": "For 0 / 100 / 200 / 500 / 1,000 adaptation rollouts: Pearson r = 0.9020 / 0.9580 / 0.9760 / 0.9890 / 0.9940; Spearman rho = 0.800 / 0.900 / 1.000 / 1.000 / 1.000; Kendall tau = 0.600 / 0.800 / 1.000 / 1.000 / 1.000; MMRV = 0.162 / 0.074 / 0.000 / 0.000 / 0.000; MAE = 8.6 / 5.3 / 4.0 / 2.9 / 2.4 percentage points (Table 11).",
    "n_policies": "5 checkpoints of one VLA training run",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2609.12036",
      "title": "Pelican-Sim 1.0: A General World Model Simulator for Embodied Intelligence (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Target is the RoboTwin simulator, not real robots. The VLA is not named in Sec. 4.6. The abstract's 'Pearson correlation of 0.994 across five checkpoints' is the 1,000-rollout budget; later experiments use the 500-rollout variant. VLM judge accuracy on generated videos 91.4% (92.6% on simulator videos), progress MAE 0.078, validity AUROC 0.953 (Table 10)."
   },
   {
    "name": "DexTouch-WM vs real dexterous-robot scores",
    "date": "2026-09-17",
    "world_model": "DexTouch-WM adapted per task: WM-Robot (200 robot trajectories) and WM-Mix (100 robot + 100 human trajectories); action-conditioned visuo-tactile video world model",
    "compared_with": "real robots",
    "comparison_detail": "Real scores of 3 policies (FTP-1, pi0.5, X-VLA) on 4 tasks (Place Shoes, Place Phone, Stack Bowls, Stand Bottle) on a Tianji arm with a 20-DoF Wuji hand; closed-loop world-model rollouts from matched initial frames; 10 matched rollouts per policy-task pair per environment, each scored by five human raters.",
    "statistic": "Per task (Place Shoes / Place Phone / Stack Bowls / Stand Bottle): WM-Robot Pearson 0.400 / 0.945 / 0.906 / 0.334, MMRV 0.033 / 0 / 0 / 0.067 (mean Pearson 0.646, mean MMRV 0.025); WM-Mix Pearson 0.972 / 0.939 / 0.889 / 0.577, MMRV 0 / 0 / 0 / 0.067 (mean Pearson 0.844, mean MMRV 0.017).",
    "n_policies": "3 policies per task, 4 tasks",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.20649",
      "title": "DexTouch-WM full text (arXiv HTML v2), Sec. IV-D, Tables II-III",
      "type": "paper",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2609.20649",
      "title": "DexTouch-WM: Learning Action-Conditioned Tactile World Models from Human Touch for Dexterous Robot Manipulation (arXiv abstract; v1 2026-09-17, v2 2026-09-18)",
      "type": "paper",
      "date": "2026-09-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Each Pearson value uses only 3 policies. Both models reverse pi0.5 and X-VLA on Stand Bottle (MMRV 0.0667). Bias from Table II: world-model scores are mostly higher than real for FTP-1 and pi0.5 (e.g. pi0.5 Place Shoes real 0.450 vs 0.950 WM-Robot and 0.900 WM-Mix) and sometimes lower for X-VLA (Stand Bottle real 0.600 vs 0.150 and 0.400). Pooled Pearson between WM-Robot and WM-Mix scores (12 pairs) is r = 0.925, which measures consistency between variants. Real trials by the authors.",
    "r_main": null
   },
   {
    "name": "Interactive World Simulator vs real-robot task scores",
    "date": "2026-03-09",
    "world_model": "Interactive World Simulator (consistency-model latent dynamics and image decoder, trained per task)",
    "compared_with": "real robots",
    "comparison_detail": "Real ALOHA task scores of DP, ACT, pi0 and pi0.5 (final and intermediate checkpoints) on 4 tasks (T pushing, rope routing, mug grasping, pile sweeping); 20 initial configurations per task sampled from the simulator's training distribution; each policy deployed in both the world model (closed-loop) and the real world.",
    "statistic": "r = 0.8553 (T Pushing), 0.8455 (Rope Routing), 0.8869 (Mug Grasping), 0.9908 (Pile Sweeping) (Fig. 7; the paper labels the coefficient r without naming it)",
    "n_policies": "final and intermediate checkpoints of 4 policy types per task (exact count not stated)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.08546",
      "title": "Interactive World Simulator full text (arXiv HTML v1), Sec. IV-A, IV-C, IV-D",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.08546v1/correlation_v3.png",
      "title": "Interactive World Simulator Fig. 7: world-simulator vs real task scores per task (r values)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://roboticsproceedings.org/rss22/p018.pdf",
      "title": "RSS 2026 paper PDF p018 (Fig. 7 r values)",
      "type": "paper",
      "date": "2026-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2603.08546",
      "title": "Interactive World Simulator for Robot Policy Training and Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real trials by the authors. Error bars are Clopper-Pearson intervals. For tasks other than T pushing, the fitted lines show a slight positive bias (higher scores in the world simulator). The same r values appear in the RSS 2026 proceedings PDF.",
    "r_main": 0.8553
   },
   {
    "name": "dWorldEval vs LIBERO simulator success (pi0 checkpoints)",
    "date": "2026-04-24",
    "world_model": "dWorldEval (discrete-diffusion world model with sparse keyframe memory and progress token)",
    "compared_with": "other",
    "comparison_detail": "LIBERO simulator success rates of pi0 checkpoints (legend: 5k, 1w, 2w, 4w, 6w training steps) on LIBERO-Object, -Spatial and -Goal; 20 episodes per task; single-view and multi-view settings.",
    "statistic": "Single-view: r = 0.860, MMRV = 0.013. Multi-view: r = 0.910, MMRV = 0.019; no-memory ablation r = 0.786, MMRV = 0.073 (Fig. 7a-b).",
    "n_policies": "5 pi0 checkpoints x 3 LIBERO suites (from the Fig. 7 legend)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2604.22152",
      "title": "dWorldEval: Scalable Robotic Policy Evaluation via Discrete Diffusion World Model (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Target is the LIBERO simulator. Training data adds 1k failed rollouts to 5.5k official demonstrations."
   },
   {
    "name": "dWorldEval vs RoboTwin simulator success",
    "date": "2026-04-24",
    "world_model": "dWorldEval (discrete-diffusion world model)",
    "compared_with": "other",
    "comparison_detail": "RoboTwin simulator success rates (ARX arm configuration) of pi0, DexVLA and Diffusion Policy on tasks shown in Fig. 7c (place empty cup, place container plate, stack bowls two; clean and randomized variants); 20 episodes per task.",
    "statistic": "r = 0.927, MMRV = 0.033; no-memory ablation r = 0.863, MMRV = 0.033 (Fig. 7c)",
    "n_policies": "3 policies (pi0, DexVLA, DP) on several RoboTwin tasks",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Target is the RoboTwin simulator. No external world-model baselines appear in this panel."
   },
   {
    "name": "dWorldEval vs real AgileX robot success",
    "date": "2026-04-24",
    "world_model": "dWorldEval (discrete-diffusion world model)",
    "compared_with": "real robots",
    "comparison_detail": "Real success rates of pi0, DexVLA and Diffusion Policy on 5 tasks (Bussing/Clean Table, Place Cup, Handover Block, Strike Block, Place Bottles) on a bimanual AgileX system with two 6-DoF arms and three RealSense cameras; 30 real episodes per task.",
    "statistic": "r = 0.918, MMRV = 0.02; no-memory ablation r = 0.829, MMRV = 0.033 (Fig. 7d)",
    "n_policies": "3 policies x 5 tasks",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real trials by the authors (Current Robotics). No external world-model baselines appear in this panel.",
    "r_main": 0.918
   },
   {
    "name": "WorldEval vs LIBERO simulator success (measured in the dWorldEval paper)",
    "date": "2026-04-24",
    "world_model": "WorldEval (video-diffusion world model), retrained by the dWorldEval authors on the same data split",
    "compared_with": "other",
    "comparison_detail": "Same LIBERO single-view setting as the dWorldEval row (pi0 checkpoints, 3 suites).",
    "statistic": "r = 0.807, MMRV = 0.024 (Fig. 7a)",
    "n_policies": "5 pi0 checkpoints x 3 LIBERO suites",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2505.19017",
      "title": "WorldEval: World Model as Real-World Robot Policies Evaluator (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Author overlap: Yaxuan Li and Yichen Zhu are authors of both WorldEval and dWorldEval. Values read from Fig. 7a; the text says baselines reach 'MMRV up to 0.039'."
   },
   {
    "name": "Ctrl-World vs LIBERO simulator success (measured in the dWorldEval paper)",
    "date": "2026-04-24",
    "world_model": "Ctrl-World (video-diffusion world model), retrained by the dWorldEval authors on the same data split",
    "compared_with": "other",
    "comparison_detail": "Same LIBERO single-view setting as the dWorldEval row.",
    "statistic": "r = 0.841, MMRV = 0.033 (Fig. 7a)",
    "n_policies": "5 pi0 checkpoints x 3 LIBERO suites",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "No author overlap between dWorldEval (Li, Zhou, Chen, Xue, Zhu) and Ctrl-World (Guo, Shi, Chen, Finn). Retrained model, not the original release."
   },
   {
    "name": "WorldGym vs LIBERO simulator success (measured in the dWorldEval paper)",
    "date": "2026-04-24",
    "world_model": "WorldGym (video-diffusion world model), retrained by the dWorldEval authors on the same data split",
    "compared_with": "other",
    "comparison_detail": "Same LIBERO single-view setting as the dWorldEval row.",
    "statistic": "r = 0.803, MMRV = 0.039 (Fig. 7a)",
    "n_policies": "5 pi0 checkpoints x 3 LIBERO suites",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2506.00613",
      "title": "WorldGym: World Model as An Environment for Policy Evaluation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-05-31",
      "accessed": "2026-10-10"
     }
    ],
    "note": "No author overlap between dWorldEval and WorldGym (Quevedo, Sharma, Sun, Suryavanshi, Liang, Yang). Retrained model, not the original release. The paper shows WorldGym turning a missed grasp into a successful pickup (Fig. 5)."
   },
   {
    "name": "GigaWorld-1 closed-loop success-rate alignment with real robots",
    "date": "2026-07-02",
    "world_model": "GigaWorld-1 (autoregressive diffusion-transformer video world model with pixel-aligned control maps and history memory)",
    "compared_with": "real robots",
    "comparison_detail": "Real-robot success rates at task and subtask level on 4 WMBench closed-loop tasks (put banana into basket; put green bowl into pink plate; fold paper boxes; pour the fries into the box), 12 subtasks; generated outcomes judged by the hybrid VLM evaluator.",
    "statistic": "No correlation statistic. Fitted line of generated vs real success rate: y = 1.134x - 0.091 (Fig. 16). Generated minus real success per subtask ranges from -0.13 to +0.07 across 12 subtasks (Fig. 17).",
    "n_policies": "not stated (WMBench protocol uses multiple policy checkpoints)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/images/multi_task_success_rate_fit.png",
      "title": "GigaWorld-1 Fig. 16: real vs generated success rate with fit lines",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/success_rate_difference_bar.svg",
      "title": "GigaWorld-1 Fig. 17: Gen - Real success-rate difference per subtask and model",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2607.02642",
      "title": "GigaWorld-1: A Roadmap to Build World Models for Robot Policy Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real executions come from the WMBench data collected by the GigaWorld-1 team (teleoperation and GigaBrain policy rollouts). Robot model not named in Sec. 6.5.5. GigaWorld-1 underestimates the last subtasks of the fold and pour tasks (-0.13 to -0.07) and slightly overestimates the first tasks (up to +0.07).",
    "r_main": null
   },
   {
    "name": "CVPR 2026 challenge world models vs real robots (measured in the GigaWorld-1 paper)",
    "date": "2026-07-02",
    "world_model": "Three anonymous world models submitted to the CVPR 2026 GigaBrain Challenge (cvpr_world_challenge_model_1-3)",
    "compared_with": "real robots",
    "comparison_detail": "Same 4 closed-loop tasks and 12 subtasks as the GigaWorld-1 row.",
    "statistic": "No correlation statistic. Fitted lines: model_1 y = 0.689x + 0.151; model_2 y = 0.719x + 0.268; model_3 y = 0.770x + 0.239 (Fig. 16). Generated minus real per subtask: model_1 from -0.33 to +0.20; model_2 from -0.02 to +0.30 (11 of 12 positive); model_3 from +0.04 to +0.27 (12 of 12 positive) (Fig. 17).",
    "n_policies": "not stated",
    "by": "independent",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/images/multi_task_success_rate_fit.png",
      "title": "GigaWorld-1 Fig. 16: real vs generated success rate with fit lines",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/success_rate_difference_bar.svg",
      "title": "GigaWorld-1 Fig. 17: Gen - Real success-rate difference per subtask and model",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": "'by' inferred: the GigaWorld-1 authors (challenge organisers) measured world models submitted by other teams; the teams are not named, so author overlap could not be checked. 'by' is unknown because the submitting teams are not named; the comparison was run by the GigaWorld-1 authors, who also organise the challenge. The paper says the challenge baselines 'often overestimate policy success'.",
    "r_main": null
   },
   {
    "name": "WMBench VLM judge vs human WMES ratings",
    "date": "2026-07-02",
    "world_model": "LoRA-tuned Qwen3-VL-8B-Instruct judge scoring rollouts generated by challenge world models",
    "compared_with": "human ratings",
    "comparison_detail": "VLM-predicted WMES (0-3) vs human-annotated WMES on 5,000+ generated videos.",
    "statistic": "Exact agreement 87.80%; adjacent agreement 99.16%; two-level errors 0.84%; MAE 0.1304; RMSE 0.3836; quadratic weighted kappa 0.7349; Spearman 0.7574; Kendall tau-b 0.7507; weighted F1 0.8744; |Bias| 0.0455 (Table 1).",
    "n_policies": "not applicable (5,000+ videos)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Validates the automatic judge, not the world models against real robots."
   },
   {
    "name": "WMBench automatic video metrics vs human WMES",
    "date": "2026-07-02",
    "world_model": "World models submitted to WMBench / the CVPR 2026 challenge",
    "compared_with": "human ratings",
    "comparison_detail": "Pearson correlation across submissions between each automatic metric (WorldArena-derived) and the human WMES score; 95% bootstrap confidence intervals with 10,000 resamples.",
    "statistic": "Group level: Visual Fidelity rho = 0.78, Geometry 0.71, Semantics 0.59. Metric level: Subject Consistency 0.88, Perspectivity 0.86, Instruction Following 0.84, Semantic Alignment 0.11, Background Consistency -0.45, Photometric Consistency -0.42, Interaction Quality -0.11.",
    "n_policies": "not stated (number of submissions)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Shows which automatic video metrics track human-judged evaluator quality; appearance-stability metrics reward static videos."
   },
   {
    "name": "EnerVerse-AC vs real-robot success (four tasks, three training steps)",
    "date": "2025-05-14",
    "world_model": "EnerVerse-AC (EVAC; action-conditioned multi-view video world model)",
    "compared_with": "real robots",
    "comparison_detail": "Real success rates of a single-view Go-1 policy (without latent planner) on 4 retrieval tasks, 40 trials per task, and of one policy at 3 training steps on Take a Bottle; three independent evaluators judged both real and EVAC videos; EVAC rollouts start from the real initial frames.",
    "statistic": "No statistic; bar charts show the same task ranking and the same training-step trend. Real vs EVAC per task: Take a Bottle 28% vs 25%; Take a Toast 100% vs 90%; Take a Bacon 85% vs 88%; Take a Leaf 55% vs 50%. Per training step (Take a Bottle): 4K 40% vs 40%; 8K 61% vs 63%; 13K 79% vs 76% (Fig. 7).",
    "n_policies": "1 policy on 4 tasks; 3 training-step checkpoints on one task",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.09723",
      "title": "EnerVerse-AC full text (arXiv HTML v1), Sec. 3 'Evaluator for Policy Model', Sec. 4.3, Appendix A.4.1",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.09723v1/Fig5_Comp_Real_CAE.svg",
      "title": "EnerVerse-AC Fig. 7: success rate per task and per learning step, real robot vs EVAC",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2505.09723",
      "title": "EnerVerse-AC: Envisioning Embodied Environments with Action Condition (arXiv abstract, v1)",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real trials by the authors (AgiBot). Absolute gaps are up to 10 percentage points (Toast). Inferred by us from the labelled values: Pearson r = 0.9858 across the 4 tasks and 0.9934 across the 3 training steps (not reported by the paper). The appendix names the third task 'ham slice' while the figure says 'Bacon'. Take a Bottle shows 28% real in the task panel but 40-79% in the training-step panel; the paper does not explain the difference.",
    "r_main": null
   },
   {
    "name": "IRASim vs LIBERO MuJoCo simulator success",
    "date": "2025-07-29",
    "world_model": "IRASim (trajectory-conditioned video diffusion transformer, OpenSora-initialised for this test)",
    "compared_with": "other",
    "comparison_detail": "LIBERO MuJoCo simulator success rates of a diffusion policy trained for 4 different numbers of steps on 'pick up the black bowl between the plate and the ramekin and place it on the plate'; 50 runs per model in each evaluator; humans judge IRASim rollouts.",
    "statistic": "Pearson correlation = 0.99. Success rates simulator vs IRASim: 0.18 vs 0.28; 0.50 vs 0.48; 0.80 vs 0.74; 1.00 vs 0.96 (Table 4).",
    "n_policies": "4 checkpoints of one policy",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2406.14540",
      "title": "IRASim full text (arXiv HTML v2), Sec. 4.2 and Table 4",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2406.14540",
      "title": "IRASim: A Fine-Grained World Model for Robot Manipulation (arXiv abstract; v1 2024-06-20, v2 2025-07-29)",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2406.14540v1",
      "title": "IRASim arXiv HTML v1 (no policy-evaluation section, no Pearson value)",
      "type": "paper",
      "date": "2024-06-20",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Target is a simulator, not real robots. Added in arXiv v2 (2025-07-29); absent from v1. Single task."
   },
   {
    "name": "OSCAR vs RoboArena real-world success",
    "date": "2026-06-03",
    "world_model": "OSCAR (Cosmos-Predict2.5-2B fine-tuned with 2D skeleton renderings as action condition); also latent-action and mesh-conditioned variants",
    "compared_with": "real robots",
    "comparison_detail": "Real RoboArena outcomes of 7 open-source DROID policies (pi0-flow, pi0-FAST, PG-flow, PG-FSQ, PG-FAST, PG-FAST+, PG-Bin) over 65 sessions; OSCAR replays each episode's recorded actions from its first frame; GPT-5 judges success.",
    "statistic": "Skeleton: MMRV 0.571 (scale 0-6), Spearman rho +0.750, Pearson r +0.852, SISR_delta 1.73 pp. Latent action: MMRV 1.429, rho +0.643, r +0.867, SISR_delta 1.98 pp. Mesh: MMRV 0.714, rho +0.679, r +0.781, SISR_delta 3.04 pp (Table 4).",
    "n_policies": 7,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.04463",
      "title": "OSCAR full text (arXiv HTML v2), Sec. 5.4 Table 4, Appendix A.10",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2606.04463",
      "title": "OSCAR: Omni-Embodiment Action-Conditioned World Model for Robotics (arXiv abstract; v1 2026-06-03, v2 2026-06-04)",
      "type": "paper",
      "date": "2026-06-03",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Real trials were run by RoboArena, not the authors. MMRV is computed on ranks (range [0,6]) and is not comparable with [0,1] MMRV elsewhere. The paper calls skeleton conditioning the strongest correlation; latent action has the higher Pearson r. GPT-5 judge calibration on 100 real clips: 78/100 agreement with human labels, specificity 0.90, recall 0.66 (under-reports success), Pearson 0.58.",
    "r_main": 0.852
   },
   {
    "name": "Cosmos-based world model vs real Bridge-setup success (NVIDIA)",
    "date": "2025-11-14",
    "world_model": "Cosmos-Predict2-2B post-trained as an action-conditioned world model (Tseng et al.)",
    "compared_with": "real robots",
    "comparison_detail": "Real success rates of OCTO-Small, OCTO-Base and OpenVLA on 4 Bridge-setup tasks (lift the pot, carrot, eggplant, cup) in the authors' replicated environment; VLM judges world-model rollouts.",
    "statistic": "Pearson = 0.687; MMRV = 0.171 (Fig. 6)",
    "n_policies": "3 policies x 4 tasks (12 points)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520v3/real_world_video_model_correlation_plot_icra.svg",
      "title": "Tseng et al. Fig. 6: policy evaluation on the Bridge setup (Cosmos and IRASim, MMRV and Pearson)",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models (arXiv abstract; v1 2025-11-14, v3 2025-12-04)",
      "type": "paper",
      "date": "2025-11-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Number of real trials per point not stated (appendix missing from arXiv HTML and PDF). Predicted success is mostly lower than actual for higher-performing points in Fig. 6.",
    "r_main": 0.687
   },
   {
    "name": "IRASim vs real Bridge-setup success (measured by NVIDIA)",
    "date": "2025-11-14",
    "world_model": "IRASim (trajectory-conditioned video diffusion), trained by Tseng et al. as baseline",
    "compared_with": "real robots",
    "comparison_detail": "Same 12 policy-task points as the Cosmos-based row.",
    "statistic": "Pearson = 0.613; MMRV = 0.611 (Fig. 6)",
    "n_policies": "3 policies x 4 tasks (12 points)",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520v3/real_world_video_model_correlation_plot_icra.svg",
      "title": "Tseng et al. Fig. 6: policy evaluation on the Bridge setup (Cosmos and IRASim, MMRV and Pearson)",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2406.14540",
      "title": "IRASim: A Fine-Grained World Model for Robot Manipulation (arXiv abstract; v1 2024-06-20, v2 2025-07-29)",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     }
    ],
    "note": "No author overlap between Tseng et al. (Tseng, Gu, Zhang, Mao, Liu, Shkurti, Yen-Chen) and IRASim (Zhu, Wu, Guo, Liu, Cheang, Kong).",
    "r_main": 0.613
   },
   {
    "name": "Cosmos-based world model vs RoboMimic simulator success (NVIDIA)",
    "date": "2025-11-14",
    "world_model": "Cosmos-Predict2-2B post-trained as an action-conditioned world model (Tseng et al.), with ablations",
    "compared_with": "other",
    "comparison_detail": "RoboMimic simulator success rates of diffusion policies (CNN and transformer variants) on Lift, Can, Square and Tool Hang.",
    "statistic": "Pearson / MMRV per task (Lift, Can, Square, Tool Hang): full model 0.879 / 0.015, 0.860 / 0.141, 0.847 / 0.165, 0.833 / 0.217; without policy rollouts in training 0.856 / 0.222, 0.744 / 0.226, 0.505 / 0.384, 0.502 / 0.288 (Table I).",
    "n_policies": "not stated in the sections read",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Target is a simulator. Table I also reports a variant without pre-trained weights (Lift 0.822 / 0.220, Can 0.716 / 0.233)."
   },
   {
    "name": "WorldArena policy evaluator: Ctrl-World vs RoboTwin simulator",
    "date": "2026-02-09",
    "world_model": "Ctrl-World (action-conditioned robot video world model, Guo et al. 2025, arXiv 2510.10125)",
    "compared_with": "other",
    "comparison_detail": "RoboTwin 2.0 simulator success rates of the same 5 pi0.5 policies (trained on 10%, 20%, 30%, 50% and 100% of the RoboTwin data); world-model success judged by a VLM from rollouts that run until 20% more frames than the ground-truth video. Simulator success rates listed in the released evaluation code: 28.60, 34.58, 37.78, 43.52, 46.80.",
    "statistic": "Pearson r = 0.986 (Figure 4); the leaderboard Policy Evaluator track lists Ctrl-World at 98.60 (r x 100)",
    "n_policies": 5,
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/simulator_vs_worldmodel_correlation.svg",
      "title": "WorldArena Figure 4: correlation of policy evaluation results from world models and the physical simulator",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/tsinghua-fib-lab/WorldArena/main/embodied_task/policy_eval_release_bundle/Policy_eval.md",
      "title": "WorldArena Track 2 policy evaluation with a VLM judge (Policy_eval.md)",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena.hf.space",
      "title": "WorldArena leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Sim-to-sim comparison: no real-robot rollouts. Figure 4 shows world-model success values of roughly 63 to 73 against simulator values of about 29 to 47 (read from the plot), and the paper states both tested world models give consistently higher success rates than the simulator, which the authors attribute to partial overfitting to successful trajectories. 'Independent' because the cited Ctrl-World authors (Y. Guo, L. X. Shi, J. Chen, C. Finn, per the WorldArena reference list) do not appear among the WorldArena authors. With 5 data points per correlation, single policies weigh heavily (our reading)."
   },
   {
    "name": "WorldArena policy evaluator: Cosmos-Predict 2.5 (action) vs RoboTwin simulator",
    "date": "2026-02-09",
    "world_model": "Cosmos-Predict 2.5, action-conditioned variant (NVIDIA video world model, post-trained on RoboTwin data by the WorldArena authors)",
    "compared_with": "other",
    "comparison_detail": "Same 5 pi0.5 policies and RoboTwin 2.0 simulator success rates as the Ctrl-World test.",
    "statistic": "Pearson r = 0.483 (Figure 4); leaderboard Policy Evaluator track lists Cosmos-Predict 2.5(action) at 48.30",
    "n_policies": 5,
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/simulator_vs_worldmodel_correlation.svg",
      "title": "WorldArena Figure 4: correlation of policy evaluation results from world models and the physical simulator",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena-worldarena.hf.space",
      "title": "WorldArena leaderboard (Gradio app, tables read through its gradio_api reload_data endpoints)",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Sim-to-sim. World-model success values in Figure 4 lie between about 46 and 52 and do not follow the simulator ordering. The leaderboard's Policy Evaluator track (19 entries, read 2026-10-10) ranges from 99.53 (WorldScape v0.2) to 7.30 (antegg); 13 entries are above 90, IRASim is 65.80 and an ACT entry is 42.72."
   },
   {
    "name": "WorldArena EWMScore vs human ratings",
    "date": "2026-02-09",
    "world_model": "14 embodied and general video world models evaluated in WorldArena",
    "compared_with": "human ratings",
    "comparison_detail": "Human evaluation scores (70 annotators, 3,500 videos; 1-5 scores on overall quality, instruction following and physical adherence normalised to 0-100) compared model by model with EWMScore.",
    "statistic": "Pearson r = 0.825 (EWMScore vs human evaluation, Figure 5); for comparison EWMScore vs data-engine performance r = 0.600 and vs action-planner performance r = 0.360 (6 models each)",
    "n_policies": "14 world models (human); 6 world models (data engine, action planner)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/ewm_human_dataengine_actionplanner_3subplots_14models.svg",
      "title": "WorldArena Figure 5: EWMScore vs human evaluation, data engine and action planner",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Agreement is computed over model-level averages. The paper does not report inter-annotator agreement."
   },
   {
    "name": "WorldArena 2.0 cross-platform agreement of world-model task success (simulators vs real robot)",
    "date": "2026-05-18",
    "world_model": "6 video world models used as data engines and action planners: GigaWorld, Genie Envisioner, TesserAct, Vidar, Wan 2.2, CogVideoX",
    "compared_with": "real robots",
    "comparison_detail": "Average task success rate of each world model (data engine and action planner) on RoboTwin 2.0 and LIBERO compared with its average success on real AgileX split-type ALOHA tasks (pour water, wipe table).",
    "statistic": "Spearman rho = 0.348 (p = 0.499) RoboTwin vs real; Spearman rho = 0.522 (p = 0.288) LIBERO vs real; Spearman rho = 0.771 (p = 0.072) RoboTwin vs LIBERO (Figure 6)",
    "n_policies": "6 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2605.17912",
      "title": "WorldArena 2.0 full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2605.17912v1/task_success_correlation.svg",
      "title": "WorldArena 2.0 Figure 6: cross-platform task success rate correlation",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://worldarena2.github.io/",
      "title": "WorldArena 2.0 project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "Real-world success rates were newly collected for the paper (Table 3; several models score 0 on real tasks). Video-quality rankings of 12 models, RoboTwin vs real data (Figure 8): visual quality rho = 0.420 (p = 0.175), motion quality 0.678 (p = 0.015), content consistency -0.273 (p = 0.391), physics adherence 0.811 (p = 0.001), 3D accuracy 0.427 (p = 0.167), controllability 0.252 (p = 0.430). Author overlap between the benchmark team and each world model's builders was not checked, so 'by' is set to authors.",
    "r_main": null
   },
   {
    "name": "DreamGen Bench automatic instruction-following vs human judgments",
    "date": "2025-05-19",
    "world_model": "4 fine-tuned video world models: Hunyuan-sft, CogVideoX-sft, WAN2.1-sft, Cosmos-sft",
    "compared_with": "human ratings",
    "comparison_detail": "Model-level IF scores from GPT-4o and from Qwen2.5-VL compared with human binary success judgments (IF-human) on each dataset.",
    "statistic": "GPT-4o vs human Pearson r: RoboCasa 0.94, GR1-Object 0.93, GR1-Behavior 0.96, GR1-Env 1.00 (Table 6). Qwen2.5-VL vs human Pearson r: RoboCasa 0.92, GR1-Object 0.95, GR1-Behavior 0.97, GR1-Env 0.96 (Table 7). Main text: average Pearson correlation above 90%.",
    "n_policies": "4 world models per dataset",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.12705v2",
      "title": "DreamGen full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Each r is computed over 4 model-level points. Inconsistencies inside the paper: Appendix H.3 says it evaluates '3 fine-tuned video world models' and calculates AUC-ROC, but Tables 6-7 list 4 models and report Pearson r; WAN2.1-sft GR1-Behavior IF-human is 74.5 in Table 6 and 70.2 in Table 7. DreamGen authors include NVIDIA Cosmos staff (for example Ming-Yu Liu), so the Cosmos comparison is not independent."
   },
   {
    "name": "DreamGen Bench score vs RoboCasa downstream policy success",
    "date": "2025-05-19",
    "world_model": "8 video world model variants (Hunyuan, CogVideoX, WAN 2.1, Cosmos; zero-shot and fine-tuned)",
    "compared_with": "other",
    "comparison_detail": "RoboCasa simulator benchmark score of a GR00T N1 policy trained only on 7K neural trajectories generated by each world model, plotted against the DreamGen Bench score (average of IF-GPT and PA).",
    "statistic": "No coefficient reported; the paper states a positive correlation and Figure 6 shows a scatter of 8 model variants with a fitted line (fine-tuned Cosmos highest at a RoboCasa score of about 21).",
    "n_policies": "8 world-model variants (one policy trained per variant)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.12705v2",
      "title": "DreamGen full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.12705v2/scatter_plot.svg",
      "title": "DreamGen Figure 6: performance correlation between DreamGen Bench and RoboCasa",
      "type": "paper",
      "date": "2025-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Simulator comparison, no real robots. The 4 zero-shot variants cluster near zero on both axes, so the visible trend is driven largely by zero-shot versus fine-tuned groups (our reading of Figure 6)."
   },
   {
    "name": "WorldSimBench Human Preference Evaluator vs human ratings",
    "date": "2024-10-23",
    "world_model": "Videos from 8 fine-tuned video generation models (Open-Sora-Plan, Lavie, ModelScope, OpenSora, AnimateDiff, DynamiCrafter, EasyAnimate)",
    "compared_with": "human ratings",
    "comparison_detail": "Agreement of the Human Preference Evaluator (HPE) and of GPT-4o with human HF-Embodied scores: accuracy for Minecraft (1-2 scale), Pearson linear correlation (PLCC) for driving and manipulation (1-5 scale); held-out tests train HPE without one generator's videos and evaluate on that generator.",
    "statistic": "Open-ended (Minecraft) accuracy: HPE 89.4 vs GPT-4o 72.8; held-out OpenSora 71.6 vs 66.5; held-out Lavie 87.9 vs 78.5. Driving PLCC: HPE 0.60 vs GPT-4o 0.28; held-out OpenSora 0.34 vs 0.03; held-out Lavie 0.49 vs -0.04. Manipulation PLCC: HPE 0.43 vs GPT-4o 0.07; held-out OpenSora 0.47 vs -0.06; held-out Lavie 0.44 vs 0.17 (Table 3; same values in the PMLR PDF).",
    "n_policies": "videos from 8 generators (validation set size not stated)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2410.18072",
      "title": "WorldSimBench full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2024-10-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/qin25f/qin25f.pdf",
      "title": "WorldSimBench PMLR PDF (affiliation footnote, Table 3)",
      "type": "paper",
      "date": "2025-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Text-table mismatch: the text says GPT-4o is negatively correlated for OpenSora in driving and Lavie in manipulation, but Table 3 shows the negative values for Lavie in driving (-0.04) and OpenSora in manipulation (-0.06). The HPE is trained on the same kind of human labels it is tested against."
   },
   {
    "name": "WoW-World-Eval overall score vs human ratings",
    "date": "2026-01-07",
    "world_model": "Image-to-video models evaluated on WoW-World-Eval (9 model variants, short and dense prompts)",
    "compared_with": "human ratings",
    "comparison_detail": "Benchmark overall score vs human overall score (sum of four 1-5 ratings: video quality, instruction understanding, physical law, planning) from 15 domain experts rating over 1,200 videos.",
    "statistic": "Overall: Pearson r = 0.93, Spearman rho = 0.91 (Figure 3b). Per dimension (Appendix 10.2): video quality r = 0.66, rho = 0.73; instruction understanding r = 0.75, rho = 0.71; physical law r = 0.81, rho = 0.83; planning r = 0.43, rho = 0.51.",
    "n_policies": "14 points in Figure 3b (labelled 'Samples'; the paper does not say what each point is)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2601.04137v1/correlation.png",
      "title": "WoW-World-Eval Figure 3b: overall score on metric vs overall score on human (r = 0.93, rho = 0.91)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The benchmark authors also built the WoW models included in the comparison."
   },
   {
    "name": "WoW-World-Eval human Turing test vs benchmark score",
    "date": "2026-01-07",
    "world_model": "Image-to-video models evaluated on WoW-World-Eval",
    "compared_with": "human ratings",
    "comparison_detail": "Deceive-human ratio (share of generated videos that 13 participants judged real in a two-alternative forced-choice test) vs benchmark scores.",
    "statistic": "Pearson r = 0.679 with the overall score; video quality r = 0.874; physical law r = 0.753",
    "n_policies": "not stated (models evaluated: 9 variants)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "name": "WoW-World-Eval benchmark score vs real-robot IDM execution success",
    "date": "2026-01-07",
    "world_model": "8 image-to-video models (Kling, Hailuo, CogVideoX, Cosmos-Predict1, Wan2.1, Cosmos-Predict2, WoW-wan, WoW-cosmos2)",
    "compared_with": "real robots",
    "comparison_detail": "Real-world success rate of actions inferred by a gripper-centric IDM (GC-IDM) from each model's generated videos on 9 manipulation tasks, compared with the model's overall benchmark score.",
    "statistic": "No coefficient reported (the paper states real-robot success 'is also correlated' with benchmark scores). Computed by us from Tables 2-3 (overall) and Table 5 (IDM success): Pearson r = 0.374, Spearman rho = 0.39 over 8 models.",
    "n_policies": "8 world models",
    "by": "authors",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Real-robot success rates were newly collected: Kling 9.88%, Hailuo 2.47%, CogVideoX 0.00%, Cosmos-Predict1 0.00%, Wan2.1 0.00%, Cosmos-Predict2 8.64%, WoW-wan 40.74%, WoW-cosmos2 18.52%. All values are multiples of 1/81, consistent with 81 trials per model (inferred; trial count not stated). The IDM itself replayed ground-truth videos with 90% overall real-robot success (10 videos per task). The top WoW scores come from the authors' own model.",
    "r_main": 0.374
   },
   {
    "name": "RoboWM-Bench real-to-sim outcome consistency",
    "date": "2026-04-21",
    "world_model": "Not a world model: checks RoboWM-Bench's reconstructed simulation scenes used to score world-model videos",
    "compared_with": "real robots",
    "comparison_detail": "For each of 7 robot tasks, 10 successful and 10 failed real-world robot trajectories were replayed with the same actions in the reconstructed simulation scene, and outcomes compared.",
    "statistic": "Success consistency 10/10 and failure consistency 10/10 for every task (Pick Object, Pull Object, Push Object, Put on Plate, Discard Trash, Close Drawer, Put in Drawer): 140 of 140 outcomes matched",
    "n_policies": "140 real-world trajectories (20 per task, 7 tasks)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.19092v2",
      "title": "RoboWM-Bench full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://robowm-bench.github.io/RoboWM-Bench/",
      "title": "RoboWM-Bench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "Real trajectories were newly collected. A second check replayed actions extracted from real successful videos in simulation: retargeting 97.1% average success (7 human tasks); IDM pretrained in simulation and fine-tuned on real data 95.7%; IDM trained on 50 real trajectories per task only 71.4%.",
    "r_main": null
   },
   {
    "name": "VideoCon-Physics vs human ratings (VideoPhy)",
    "date": "2024-06-05",
    "world_model": "VideoCon-Physics, automatic judge of semantic adherence and physical commonsense for text-to-video outputs",
    "compared_with": "human ratings",
    "comparison_detail": "Majority-vote binary SA and PC labels from 3 Amazon Mechanical Turk raters on videos for the 344 test prompts; the judge was trained on human labels for the 344 training prompts from 9 models",
    "statistic": "ROC-AUC (x100) on test prompts: SA 82, PC 73 (zero-shot VideoCon 65 and 54; Gemini-1.5-Pro-Vision 73 and 58 in Table 4, while the text gives 54 for PC; GPT-4-Vision 53 and 53). On three generators unseen in training: SA 79, PC 72 (VideoCon 64 and 57). Human inter-annotator agreement: 75% SA, 70% PC.",
    "n_policies": "12 text-to-video models (videos for 344 test prompts)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2406.03520",
      "title": "VideoPhy (arXiv HTML v2), Tables 4, 5 and 10",
      "type": "paper",
      "date": "2024-10-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Hritikbansal/videophy",
      "title": "VideoPhy README leaderboards",
      "type": "repo",
      "date": "2026-01-30",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Model ranking: automatic order of open models CogVideoX-5B > VideoCrafter2 > LaVIE > CogVideoX-2B > SVD-T2I2V > ZeroScope > OpenSora versus human order CogVideoX-5B > VideoCrafter2 > CogVideoX-2B > LaVIE > SVD-T2I2V > ZeroScope > OpenSora. The paper notes Pika scores low on the automatic leaderboard; in the README Pika is 2nd of 12 on the human leaderboard and 12th of 14 on the automatic one. The human labels were newly collected for this paper."
   },
   {
    "name": "VideoPhy-2-Autoeval vs human ratings",
    "date": "2025-03-09",
    "world_model": "VideoPhy-2-Autoeval, 7B automatic judge for text-to-video outputs (SA, PC, rule grounding)",
    "compared_with": "human ratings",
    "comparison_detail": "Averaged 1-5 SA and PC scores and majority-vote rule labels from 3 AMT raters on the 590-prompt test split, for (a) unseen prompts from the training models and (b) unseen video models",
    "statistic": "Pearson r (x100) between predicted and human 1-5 scores: unseen prompts SA 47.0, PC 37.0 (average 42.0); unseen video models SA 45.0, PC 37.0 (average 41.0). Baselines on PC: Gemini-2.0-Flash-Exp 11.0 and 11.0; VideoCon-Physics 25.0 and 26.0; VideoScore 10.0 and 13.0. Joint score (SA >= 4 and PC >= 4) accuracy / F1: 79.1 / 51.1 (unseen prompts), 76.3 / 49.3 (unseen models). Rule classification accuracy: 78.7 and 72.9 (Gemini-2.0-Flash-Exp 59.2 and 57.1). Human inter-annotator agreement 75% to 80%.",
    "n_policies": "7 text-to-video models; 590 test prompts",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2503.06800",
      "title": "VideoPhy-2 (arXiv HTML v1), Tables 4-6",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The judge was trained on about 50K human annotations of videos from HunyuanVideo-13B, Cosmos-Diffusion-7B and CogVideoX-5B on the 3,350 training prompts. An independent test of this judge on real recordings is listed separately (Morpheus)."
   },
   {
    "name": "PhyGenEval vs human ratings (PhyGenBench)",
    "date": "2024-10-07",
    "world_model": "PhyGenEval, hierarchical VLM/LLM judge (VQAScore, GPT-4o, LLaVA-Interleave, InternVideo2) for text-to-video physical commonsense",
    "compared_with": "human ratings",
    "comparison_detail": "64 randomly selected prompts x 8 models = 512 videos; 3 annotators give 0-3 physical-correctness scores, averaged and rounded up",
    "statistic": "Overall PCA correlation with humans: Kendall tau = 0.78, Spearman rho = 0.81. Per domain (tau / rho): mechanics 0.72 / 0.75, optics 0.76 / 0.77, thermal 0.73 / 0.75, material 0.81 / 0.84. Other evaluators overall (tau / rho): VideoScore 0.17 / 0.19, DEVIL 0.17 / 0.18, VideoPhy (VideoCon-Physics) 0.03 / 0.04.",
    "n_policies": "512 videos from 8 text-to-video models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2410.05363",
      "title": "PhyGenBench paper (arXiv HTML v1), Table 1",
      "type": "paper",
      "date": "2024-10-07",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Correlations are at the video level. For semantic alignment, Appendix results report GPT-4o-based PhyGenEval at Kendall tau 0.53 / Spearman rho 0.56 versus 0.42 / 0.44 with LLaVA. The GPT-4o-written questions and standards are fixed per prompt."
   },
   {
    "name": "VBench dimensions vs human preference",
    "date": "2023-11-29",
    "world_model": "VBench automatic dimension metrics for text-to-video quality",
    "compared_with": "human ratings",
    "comparison_detail": "Pairwise human preferences per dimension (N prompts x 5 groups x 6 pairs), converted to model win ratios and compared with win ratios computed from VBench scores",
    "statistic": "Per-dimension correlation of win ratios, labelled Spearman's rho (Table A5): subject consistency 96.51%, background consistency 94.80%, temporal flickering 88.73%, motion smoothness 99.80%, dynamic degree 82.09%, aesthetic quality 98.65%, imaging quality 92.16%, object class 80.37%, multiple objects 98.98%, human action 89.15%, color 60.73%, spatial relationship 97.59%, scene 94.07%, appearance style 99.65%, temporal style 97.53%, overall consistency 93.27%.",
    "n_policies": "4 text-to-video models (LaVie, ModelScope, VideoCrafter, CogVideo)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2311.17982",
      "title": "VBench (arXiv HTML v1), Figure 5 and Table A5",
      "type": "paper",
      "date": "2023-11-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2411.13503",
      "title": "VBench++ (arXiv HTML v1), human alignment section",
      "type": "paper",
      "date": "2024-11-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Each correlation rests on 4 data points (one per model). Inferred: a Spearman rank correlation over 4 points can only take values in steps of 0.2, so values such as 96.51% indicate a different computation (likely Pearson) despite the label. None of the dimensions measures physical correctness. VBench++ adds human annotations for image-to-video and trustworthiness dimensions (values in figures, not extracted)."
   },
   {
    "name": "VBench-2.0 dimensions vs human preference",
    "date": "2025-03-27",
    "world_model": "VBench-2.0 automatic evaluators (VLM/LLM question answering and specialist models) for intrinsic faithfulness",
    "compared_with": "human ratings",
    "comparison_detail": "Pairwise human preferences (group comparisons for diversity) per dimension, converted to model win ratios; 284 annotation hours across 18 annotators; 20% of pairs re-checked with a 95% pass requirement",
    "statistic": "Per-dimension correlation of VBench-2.0 and human win ratios (Table IV; Figure 11 labels it Spearman's rho): physics dimensions mechanics 93.56%, thermotics 93.86%, material 93.71%, multi-view consistency 98.44%; commonsense dimensions motion rationality 87.97%, instance preservation 99.11%; range across all 18 dimensions 81.70% (composition) to 99.46% (human identity).",
    "n_policies": "4 text-to-video models (HunyuanVideo, CogVideoX-1.5, Sora, Kling 1.6)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2503.21755",
      "title": "VBench-2.0 (arXiv HTML v2), Table IV and Figure 11",
      "type": "paper",
      "date": "2025-08-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Same caveat as VBench: each correlation uses 4 model-level points, and the reported values are not possible for a 4-point Spearman correlation (inferred). The authors state that current large models cannot serve as reliable evaluators for human anatomy and motion rationality without specialist models."
   },
   {
    "name": "WorldModelBench judge vs human votes",
    "date": "2025-02-28",
    "world_model": "Fine-tuned 2B VLM judge (VILA family) for video generators as world models",
    "compared_with": "human ratings",
    "comparison_detail": "Crowd votes on 8 criteria per video; per-criterion error on held-out videos (713 videos from 2 held-out models per prompt) and total-score comparison per model across all 350 prompts",
    "statistic": "Mean relative error of judge total score versus human total score: 4.1% across 14 models (maximum 6.81%, runway); the judge scores 13 of 14 models higher than humans do. Per-criterion prediction error on held-out videos (instruction following / commonsense / physics adherence): VILA-2B + CoT fine-tuned 32.3% / 16.4% / 29.7%; row labelled VILA-2B + zero-shot 21.0% / 28.0% / 24.0%; GPT-4o 29.3% / 35.0% / 36.0%; Gemini-1.5-Pro 30.7% / 34.5% / 29.3%. Human vote agreement: 70.0% pairwise agreement; 87.1% of votes within 2 points; 96.2% (experts) and 95.4% (crowd) within one standard deviation of the expert mean.",
    "n_policies": 14,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2502.20694",
      "title": "WorldModelBench (arXiv HTML v1), Tables 2, 4 and 5",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Inferred by us from Table 4: the judge's model ranking matches the human ranking except two adjacent swaps (Spearman rho = 0.991, Kendall tau = 0.956 over 14 models). Inferred caveat: the training split takes, for each prompt, videos from 12 of the 14 models, so the per-model total-score comparison in Table 4 is largely in-sample. The headline claims ('8.6% higher average accuracy' in the abstract, '9.9% lower error rate' in the introduction) could not be reproduced exactly from Table 5."
   },
   {
    "name": "WorldScore metrics vs human preference",
    "date": "2025-04-01",
    "world_model": "WorldScore automatic metrics for 3D, 4D and video world generators",
    "compared_with": "human ratings",
    "comparison_detail": "Two-alternative forced-choice preferences from 400 participants on video pairs from CogVideoX-I2V, VideoCrafter1-I2V, DynamiCrafter, WonderJourney and InvisibleStitch; agreement score = share of participants preferring the video the metric ranks higher",
    "statistic": "Subjective quality agreement: 0.637 for the chosen mean of CLIP-IQA+ and CLIP Aesthetic (CLIP Aesthetic alone 0.628, MUSIQ 0.530; upper bound 0.772). Bucket tests (60±5 over 30±5 / 90±5 over 60±5): camera control 71.2% / 73.5%, object control 96.3% / 87.7%, 3D consistency 91.7% / 97.3%, photometric consistency 91.6% / 95.1%, motion magnitude 91.8% / 76.2%.",
    "n_policies": "videos from 5 models; 400 participants",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2504.00983",
      "title": "WorldScore (arXiv HTML v2), Appendix D, Tables S5 and S6",
      "type": "paper",
      "date": "2025-11-29",
      "accessed": "2026-10-10"
     }
    ],
    "note": "No model-level correlation between WorldScore totals and human rankings is reported. Participants answered a single question about overall quality. None of the validated metrics targets physical correctness."
   },
   {
    "name": "PhyWorldBench CAP judge vs human ratings",
    "date": "2025-07-17",
    "world_model": "Context-Aware Prompt (CAP) zero-shot MLLM judge (GPT-o1) for text-to-video physical realism",
    "compared_with": "human ratings",
    "comparison_detail": "Majority-vote yes/no SA and PC labels from 3 AMT raters on PhyWorldBench videos; 8 frames sampled per video",
    "statistic": "ROC-AUC (x100): CAP SA 80.3, PC 75.1; plain GPT-o1 75.4 / 61.6; GPT-4o 72.1 / 60.1; Gemini-2.0-Flash 74.6 / 60.9; Qwen-VL-2.0 72.4 / 59.8; CAP without context 77.3 / 65.6; CAP without chain of thought 76.3 / 73.6.",
    "n_policies": "videos from 12 text-to-video models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2507.13428",
      "title": "PhyWorldBench (arXiv HTML), Table 2, Appendix D and Appendix P",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Human and CAP leaderboards agree except second place among proprietary models (Sora for humans, Kling for CAP); the authors attribute this to Kling's cinematic style biasing CAP. The appendix reports that CAP increases false positives by 0.001 (SA) and 0.014 (PC)."
   },
   {
    "name": "IPV-Bench automatic evaluation vs human annotation",
    "date": "2025-03-18",
    "world_model": "VBench-factor visual-quality score plus GPT-4o prompt-following judge for impossible-video generation",
    "compared_with": "human ratings",
    "comparison_detail": "Model-level automatic scores versus human binary labels for visual quality, impossible-prompt following and IPV-Score",
    "statistic": "Spearman rho between automatic and human scores is above 0.8 for visual quality, prompt following and IPV-Score (exact values only in Figure 10). GPT-4o prompt-following 'human alignment' 0.72, rising to 0.80 with the authors' three-step prompt (Kling videos).",
    "n_policies": 10,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2503.14378",
      "title": "Impossible Videos (arXiv HTML v1), Appendix B.2 and Table 6",
      "type": "paper",
      "date": "2025-03-18",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The agreement measure behind 0.72 and 0.80 is not defined in the text read. Correlation over 10 models."
   },
   {
    "name": "Morpheus physics scores vs human ratings",
    "date": "2025-04-03",
    "world_model": "Morpheus physics-informed scores (dynamical and physical-invariance) for video generators on Newtonian experiments",
    "compared_with": "human ratings",
    "comparison_detail": "Three judges rated 65 generated videos for physical plausibility (1-5); alignment measured on the 42 videos kept as valid by both the automatic discard filter and the human majority",
    "statistic": "Spearman rho = 0.708 between total Morpheus score and mean human rating; dynamical score rho = 0.768; physical-invariance score rho = -0.157 (p = 0.32, not significant). Discard filter vs human majority on 65 videos: accuracy 0.785, precision 0.750, recall 0.450, F1 0.563.",
    "n_policies": "42 videos from several generators (e.g. COSMOS-predict2, Kling, Veo3)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2504.02918",
      "title": "Morpheus (arXiv HTML v3), Section 4 and Appendix D.10",
      "type": "paper",
      "date": "2026-06-29",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The authors validate the invariance score against physics instead (real videos score at least 0.90) and argue that human raters cannot see conservation violations. Reported in v3; not checked whether v1 contained it."
   },
   {
    "name": "VideoPhy-2-AutoEval on real recordings (Morpheus test)",
    "date": "2025-04-03",
    "world_model": "VideoPhy-2-AutoEval, VLM judge of physical commonsense (built by the VideoPhy-2 team)",
    "compared_with": "other",
    "comparison_detail": "Scores given to clean real-world recordings of Newtonian experiments, which should be near the ceiling, compared with scores given to generations from nine models conditioned on the same real frames",
    "statistic": "Mean PC (1-5): 3.71 for real videos vs 3.53 for generated; rule adherence 58.5% vs 51.1% (no significant gap); joint physical consistency pass rate (PC >= 4 and rule followed) 23.6% real vs 13.9% generated; 0% pass rate for real projectiles, bouncing balls, sliding books, collisions and rolling cans; rule adherence for sliding books 82.8% generated vs 40.0% real.",
    "n_policies": "real recordings plus generations from 9 models",
    "by": "independent",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2504.02918",
      "title": "Morpheus (arXiv HTML v3), Section 4 and Appendix D.6",
      "type": "paper",
      "date": "2026-06-29",
      "accessed": "2026-10-11"
     }
    ],
    "note": "No author overlap between the Morpheus paper (Tragoudaras, Zhang, Cherniavskii, Vozikis, Nijdam, Prinzhorn, Bodracska, Sebe, Zadaianchuk, Gavves) and the VideoPhy-2 paper (Bansal, Peng, Bitton, Goldenberg, Grover, Chang). 95% confidence intervals computed by Wald or normal approximation. Generations scored here used the original real frames, not the style-augmented split."
   },
   {
    "name": "Physics-IQ Verified vs original Physics-IQ rankings",
    "date": "2026-06-17",
    "world_model": "Physics-IQ scoring of image-to-video generators (original vs verified protocol)",
    "compared_with": "other",
    "comparison_detail": "Model ranking under the original protocol versus the verified protocol (corrected prompts, artifact-cleaned ground truth, per-sample score), 4 runs of 198 videos per model and setting",
    "statistic": "Spearman rho = 0.65, Kendall tau = 0.46 between the two rankings; bootstrap over 500 resampled video sets: mean rho = 0.697, mean tau = 0.513, while within-protocol bootstrap correlations exceed 0.9 with non-overlapping 95% intervals.",
    "n_policies": "6 image-to-video models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1), Section 4.1 and Appendix E.3",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Not a human-agreement study. Two Physics-IQ authors (Jaini, Geirhos) co-authored the audit. Wan 2.2 moves from first to third, P-Video from fourth to last; changing only the score formula did not change the ranking (bootstrap rho and tau about 1)."
   },
   {
    "name": "WorldLens-Agent vs human ratings",
    "date": "2025-12-11",
    "world_model": "Evaluator: WorldLens-Agent (Qwen3-VL-8B, LoRA fine-tuned on WorldLens-26K human scores); videos scored: Gen3C (zero-shot test), with qualitative examples on Cosmos-Drive and CARLA videos",
    "compared_with": "human ratings",
    "comparison_detail": "Agent scores (1 to 10) vs human annotator scores on the four human-preference dimensions for videos generated by Gen3C, a model not in the training data.",
    "statistic": "None reported. The paper states the agent's predicted scores 'exhibit strong alignment with human annotations' and shows examples (Figure 8); no correlation, error or accuracy figure appears in v1 or v2.",
    "n_policies": "not stated (number of test videos not given)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML v2, Section 5.2 and Section 12",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10958v1",
      "title": "WorldLens arXiv HTML v1",
      "type": "paper",
      "date": "2025-12-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/worldbench/WorldLens",
      "title": "WorldLens README (WorldLens-26K and Agent 'To be updated')",
      "type": "repo",
      "date": "2026-01-18",
      "accessed": "2026-10-10"
     }
    ],
    "note": "WorldLens-26K (26,808 scoring records with text rationales) and the agent were not released as of the README checked on 2026-10-10. Within the paper, text and tables differ slightly on human scores: the text says DiST-4D leads behavioral safety (2.59), but Table 29 lists OpenDWM 2.598 and DiST-4D 2.591; the text says OpenDWM has the highest realism (2.76), which matches Vehicle Realism (2.757), while DiST-4D leads Overall Realism (2.320)."
   },
   {
    "name": "ACT-Estimator vs nuScenes ground truth",
    "date": "2024-12-06",
    "world_model": "Evaluator for action-conditioned driving video models (Vista, Terra); the validation itself uses real nuScenes video",
    "compared_with": "other",
    "comparison_detail": "Real-world driving logs: ACT-Estimator predictions on 8,407 held-out real nuScenes front-camera clips (4 s) compared with action labels derived by rules from ego_pose logs and with ego trajectories from ego_pose; trajectory baseline is DROID-SLAM combined with Metric3D depth.",
    "statistic": "High-level action classification accuracy 94.03%. Trajectory error averaged over action categories: ADE 0.81 (ACT-Estimator) vs 7.52 (DROID-SLAM); FDE 1.59 vs 13.75.",
    "n_policies": "not applicable (validates the evaluator; no policies)",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2412.05337",
      "title": "ACT-Bench arXiv HTML v1, Section 4 (Figure 4, Table 3)",
      "type": "paper",
      "date": "2024-12-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The estimator is validated on real video only; its accuracy on generated video, which is what the benchmark scores, is not measured. The validation classes differ from the benchmark classes (shifting left/right excluded; constant speed split into two classes). ADE/FDE units are not stated in the table caption."
   },
   {
    "name": "DriveArena: UniAD on real vs regenerated nuScenes",
    "date": "2024-08-01",
    "world_model": "World Dreamer (DriveArena's layout-conditioned, autoregressive multi-view diffusion image generator)",
    "compared_with": "other",
    "comparison_detail": "Real-world driving logs: UniAD run on the original nuScenes validation images vs on World Dreamer images generated from the same scene layouts with the same ground-truth ego trajectories (open loop, 150 scenes at 2 Hz).",
    "statistic": "PDMS 0.910 +/- 0.09 on original images vs 0.902 +/- 0.09 on generated images (less than 1% drop). Sub-scores original vs generated: NC 0.993 vs 0.993, DAC 0.995 vs 0.991, EP 0.914 vs 0.909, TTC 0.947 vs 0.951, comfort 0.848 vs 0.821. UniAD perception and planning (Table 1), original vs generated: detection mAP 37.98 vs 16.06, NDS 49.85 vs 30.03, average planning L2 1.05 m vs 1.18 m, average collision rate 0.29% vs 0.24%.",
    "n_policies": 1,
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2408.00415v1",
      "title": "DriveArena arXiv PDF v1, Tables 1 and 2, Sections 4.3 and 4.5",
      "type": "paper",
      "date": "2024-08-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "One policy only, so no ranking agreement can be measured. The authors attribute the small PDMS gap partly to UniAD's heavy reliance on ego-state inputs, so the planning score may say little about the images. In DriveArena's own open-loop simulation (new routes and traffic) UniAD's PDMS falls to 0.636; no real-world reference exists for those routes. No comparison with CARLA results is reported, although one map replicates CARLA Town05."
   },
   {
    "name": "GameWorld Score vs human preference (Matrix-Game paper)",
    "date": "2025-06-23",
    "world_model": "Matrix-Game, Oasis and MineWorld (action-conditioned Minecraft video world models)",
    "compared_with": "human ratings",
    "comparison_detail": "Two independent double-blind annotator groups compared the three models' videos on Overall Quality, Controllability, Visual Quality and Temporal Consistency; win rate is the share of scenario-metric pairs in which a model was rated best.",
    "statistic": "No correlation statistic. Matrix-Game win rates: 96.30% Overall Quality, 93.76% Controllability, 98.23% Visual Quality, 89.56% Temporal Consistency. Matrix-Game also has the top or tied-top value on every GameWorld Score dimension; the paper describes the two as aligned.",
    "n_policies": "3 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.18701",
      "title": "Matrix-Game arXiv HTML v1, Section 6.1 and Figure 8",
      "type": "paper",
      "date": "2025-06-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The agreement rests on one model winning both comparisons; with three models it does not test whether the metrics order the other two models the way humans do. Number of annotators and videos not stated in the text. The benchmark authors also built Matrix-Game."
   },
   {
    "name": "DrivingGen metrics vs human preference",
    "date": "2026-01-04",
    "world_model": "14 video world models (general video, physical-world and driving-specific)",
    "compared_with": "human ratings",
    "comparison_detail": "Per-model win ratios from human pairwise comparisons (VBench protocol: win = 1, tie = 0.5 each) vs per-model win ratios from the primary metric of each category, over the 14 benchmarked models.",
    "statistic": "Spearman rho: video realism (FVD) 0.8899; video quality (subjective image quality) 0.8344; video consistency 0.8757; trajectory realism (FTD) 0.7682; trajectory quality 0.6261; trajectory consistency 0.7467.",
    "n_policies": "14 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML v2, Section 4.1, Appendix B.9 and Figure 5",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Values read from the Figure 5 image (arXiv v2; the appendix also exists in v1). Appendix B.9 does not state the number of annotators or comparisons. The authors attribute lower agreement for trajectory metrics to noisy monocular SLAM and metric-depth recovery on generated videos with artifacts."
   },
   {
    "name": "WorldMark v1 metrics vs human rankings",
    "date": "2026-04-23",
    "world_model": "Six interactive image-to-video world models (Yume 1.5, Matrix-Game 2.0, HY-World 1.5, HY-GameCraft, Open-Oasis, Genie 3)",
    "compared_with": "human ratings",
    "comparison_detail": "20 volunteers ranked 50 sets of first-person videos from the six models; rankings compared with automated WorldMark v1 scores.",
    "statistic": "Spearman rank correlation reported as rho > 0.9 (exact values only in a figure).",
    "n_policies": "6 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.21686v1",
      "title": "WorldMark arXiv HTML v1, Human Preference Alignment",
      "type": "paper",
      "date": "2026-04-23",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Superseded by the v2 study below, which uses different metrics and models."
   },
   {
    "name": "WorldMark v2 metrics vs human pairwise judgements",
    "date": "2026-08-05",
    "world_model": "Ten interactive image-to-video world models (pairs drawn from their outputs)",
    "compared_with": "human ratings",
    "comparison_detail": "40 annotators not involved in metric development each judged 20 blinded pairs for each of 10 dimensions (the four action metrics per axis, plus Local and Revisit Memory), about 200 judgements per annotator; pairs were first-person cases that the metric separated by more than 10 points. Global Memory and Visual Quality were not tested.",
    "statistic": "Agreement (share of trials where metric and annotator preferred the same video, counting trials with a stated preference) averaged over the ten dimensions: 85.1%, ranging from 79.1% (rotational Motion Stability) to 93.3% (Revisit Memory). Tie rate 13.4% on rotation vs 9.6% on translation for the action metrics.",
    "n_policies": "10 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.21686v2",
      "title": "WorldMark arXiv HTML v2, Section 4.3, Figure 5, Appendix G",
      "type": "paper",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     }
    ],
    "note": "This tests pairwise ordering of clearly separated pairs (gap above 10 points), which is easier than ranking close models. One evaluated model (AlayaWorld) shares authors with the benchmark."
   },
   {
    "name": "WBench metrics vs human win rates",
    "date": "2026-05-25",
    "world_model": "Interactive video world models sampled from the 20 benchmarked models",
    "compared_with": "human ratings",
    "comparison_detail": "Per-model human win rates from blind pairwise comparisons (A wins, B wins or tie; tie = 0.5) vs WBench automated scores, following the VBench protocol. 400 crowdsourced annotators; 13,515 pairwise tasks, each judged by three annotators with majority vote, plus gold-standard checks and expert sampling review.",
    "statistic": "Spearman rho >= 0.94 for all ten evaluation aspects; rho = 1.00 for event editing, subject action, perspective switching and spatial consistency.",
    "n_policies": "4 to 6 world models per aspect (main text); Appendix D.2 also mentions 'ten sampled models' and '8 evaluated models'",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2605.25874v1",
      "title": "WBench arXiv HTML v1, Section 5.4, Figure 5 and Appendix D.2",
      "type": "paper",
      "date": "2026-05-25",
      "accessed": "2026-10-11"
     }
    ],
    "note": "With 4 to 6 models per aspect a rank correlation can take few values; rho = 1.00 means the humans and the metric put those few models in the same order (inferred). The model counts for the human study are inconsistent within the paper (see n_policies). Per-aspect values other than the four at 1.00 are only in the Figure 5 SVG, whose text is outlined and was not read."
   },
   {
    "name": "PlayWorld VQA rubric scores vs human preference",
    "date": "2026-08-13",
    "world_model": "Nine interactive video world models (Genie 3, LingBot-World, LingBot-World2, HY-World2, HappyOyster, SANA-WM, Hunyuan-GameCraft, HY-WorldPlay, Matrix-Game-3.0)",
    "compared_with": "human ratings",
    "comparison_detail": "600 valid human pairwise judgements on video pairs sampled evenly across the four dimensions; wins aggregated into normalized per-model preference scores and compared with the Gemini 3.1 Pro rubric scores.",
    "statistic": "Spearman rho across the nine models: geometry consistency 0.983, interaction fidelity 0.933, out-of-sight evolution 0.812, insight evolution 0.745, overall 0.933.",
    "n_policies": "9 world models",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2608.13552v2",
      "title": "PlayWorld arXiv HTML v2, Section 4.4 and Figure 6",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Values read from the Figure 6 image. The figure legend names Hunyuan-GameCraft-2 while the text lists Hunyuan-GameCraft (Li et al., 2025). Agreement is lowest for the two state-evolution dimensions, which are also the dimensions where all models score lowest."
   },
   {
    "name": "Wayve GAIA-3 vs on-road experiments",
    "date": "2025-12-02",
    "world_model": "GAIA-3 (Wayve generative driving world model)",
    "compared_with": "other",
    "comparison_detail": "Relative performance of driving policies evaluated in GAIA-3 vs on-road experiments (cars, not robot arms).",
    "statistic": "No statistic published. The official page says correlation studies against on-road experiments indicate the model can 'reliably predict relative policy performance', without a coefficient, sample size or method.",
    "n_policies": "not stated",
    "by": "authors",
    "level": "verified",
    "sources": [
     {
      "url": "https://wayve.ai/thinking/gaia-3/",
      "title": "GAIA-3: Scaling World Models to Power Safety and Evaluation",
      "type": "blog",
      "date": "2025-12-02",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Read through WebFetch (curl returned HTTP 403). Waymo's World Model post (2026-02-06) and NVIDIA's Alpamayo-R1 paper also publish no world-model-vs-real agreement figure."
   }
  ],
  "problems": [
   {
    "text": "The same world model gets very different agreement scores depending on who tests it and where. Ctrl-World's agreement with ground truth was Pearson r = 0.53 / MMRV 0.22 in the PolaRiS real-robot study, r = 0.552 / MMRV 0.215 in the WEAVER real-robot study, r = 0.796 / MMRV 0.053 in the PersistWorld real-robot study, r = 0.841 against the LIBERO simulator in the dWorldEval study, and r = 0.986 against the RoboTwin simulator in WorldArena. Its own paper reports no coefficient.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS",
      "type": "paper",
      "date": "2025-12-18",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2606.13672",
      "title": "WEAVER",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2603.25685",
      "title": "Persistent Robot World Models (PersistWorld)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2604.22152",
      "title": "dWorldEval",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena",
      "type": "paper",
      "date": "2026-02-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2510.10125",
      "title": "Ctrl-World",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Compiled by us from the validity_studies rows; setups, tasks, policies and model versions differ (dWorldEval retrained the model), so the numbers are not directly comparable. PolaRiS shares an author (Chelsea Finn) with Ctrl-World; the other four measurers do not.",
    "featured": 2
   },
   {
    "text": "Almost every published agreement number between a world-model evaluator and real robots was measured by the world model's own builders, on 1 to 18 policies or checkpoints, often 3 to 8. Third-party checks without shared authors exist only for Ctrl-World (PersistWorld and WEAVER papers on real robots; dWorldEval and WorldArena against simulators), WorldGym (dWorldEval, simulator only), Cosmos-Predict 2.5 (WorldArena, simulator only) and IRASim (NVIDIA, real robots). Several studies reuse real-robot numbers from earlier work instead of running new paired trials (WorldGym from OpenVLA, RoboWorld and Runway from RoboArena).",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models",
      "type": "paper",
      "date": "2025-11-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Compiled by us from the validity_studies rows in this file.",
    "featured": 1
   },
   {
    "text": "The evaluators that third parties cannot run are the ones built by companies. Google DeepMind's Veo (Robotics) evaluator, Runway's GWM-Robotics, 1X's world model and Wayve's GAIA-3 have no released code or weights for evaluation, so their agreement numbers cannot be reproduced; RoboWorld's code was marked 'coming soon' on its project page.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://veo-robotics.github.io",
      "title": "Veo Robotics project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     },
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Runway GWM-Robotics post",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model report",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://byeongguks.github.io/RoboWorld/",
      "title": "RoboWorld project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "note": "From the licence fields of the protocol rows in this file.",
    "featured": 3
   },
   {
    "text": "Automated video-generation scores bunch up as models improve. NVIDIA reports that on the same prompts its evaluated text-to-video models spread over about 10 points in human evaluation but about 4 points on the automated PAIBench-G overall score, and says such metrics are usually blind to temporal and physics failures.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "report",
      "date": "2026-06-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Statement by a model developer about a benchmark it reports on; NVIDIA uses it to motivate its own human-evaluation protocols."
   },
   {
    "text": "Judge-model choice changes results. NVIDIA reports that the public PAIBench-G image-to-video leaderboard results, judged by Qwen3-VL-235B-A22B, were not reproducible, and it switched to Qwen2.5-VL-72B-Instruct as the judge for its own evaluation.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "report",
      "date": "2026-06-23",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Footnote in Section 6.2.2. The PAI-Bench authors' response was not found."
   },
   {
    "text": "Generated videos can outscore real videos on automated benchmarks. On the PAI-Bench generation leaderboard, Cosmos3-Super (83.9) and Cosmos3-Nano (83.7) score above the real source videos (82.6) overall, and in the robot domain Cosmos3-Super (90.0), Cosmos3-Nano (90.2) and Veo-3 (86.9) score above the source videos (86.2).",
    "level": "verified",
    "sources": [
     {
      "url": "https://huggingface.co/spaces/shi-labs/physical-ai-bench-leaderboard/resolve/main/data/generation-leaderboard.json",
      "title": "PAI-Bench generation leaderboard data file",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "A score above the real reference means the benchmark cannot separate the best models from reality on these measures.",
    "featured": 7
   },
   {
    "text": "Validity numbers are restated loosely in model reports. NVIDIA's Cosmos 3 report says RBench agrees with human judgments at Spearman rho = 0.96 across 25 evaluated models; the RBench paper reports that value for a ten-model subset.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.02800v4",
      "title": "Cosmos 3: Omnimodal World Models for Physical AI",
      "type": "report",
      "date": "2026-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2601.15282v1",
      "title": "Rethinking Video Generation Model for the Embodied World (RBench)",
      "type": "paper",
      "date": "2026-01-21",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Vision-language judges, which several world-model evaluators use to decide whether a task succeeded, are error-prone. On FailBench (2,197 manipulation attempts from 14 public sources, 12 real and 2 simulated), the best of 13 VLM-based detectors reached 0.77 mean balanced accuracy, fell below 0.60 on contact-intensive assembly, and leaned toward predicting success when evidence was ambiguous.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2609.03611",
      "title": "FailBench: How Reliable are VLMs at Judging Robot Task Success?",
      "type": "paper",
      "date": "2026-09-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "FailBench tests judges on recorded robot videos, not on world-model output, so the effect on world-model evaluators is inferred.",
    "featured": 5
   },
   {
    "text": "Most world-model benchmarks never execute what the model predicts. A 2026 survey of 160 benchmarks counts 34 world-model-evaluation benchmarks that all score open-loop, and only four benchmarks (RoboWM-Bench, World-in-World, WorldArena, WorldSimBench) that turn predictions into executed actions; only 11 of 160 compare a world-model approach with a direct VLA policy.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.29669",
      "title": "Do World Models Make Better Robots? A Survey of Evaluation Benchmarks for Predictive Embodied Intelligence",
      "type": "survey",
      "date": "2026-08-30",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The survey's corpus counts benchmarks, not policy-evaluation methods; world-model policy evaluators such as WorldEval, WorldGym, Ctrl-World or RoboWorld are not in its four-benchmark count."
   },
   {
    "text": "There is no agreed metric for embodied world models. A 2026 survey of world models for robot learning states that the field lacks a widely accepted evaluation metric and that comparisons are fragmented across benchmarks and protocols.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2605.00080",
      "title": "World Model for Robot Learning: A Comprehensive Survey",
      "type": "survey",
      "date": "2026-04-30",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "The usual agreement statistics have known limits. The SIMPLER authors note that Pearson r only measures a linear fit and swings with small real-world differences when policies perform similarly, and that rank correlation ignores how far apart misranked policies are; they proposed MMRV (Mean Maximum Rank Violation, 0 to 1, lower is better) to weight rank errors by the real success-rate gap.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2405.05941",
      "title": "Evaluating Real-World Robot Manipulation Policies in Simulation (SIMPLER)",
      "type": "paper",
      "date": "2024-05-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Most world-model validity studies report Pearson r and MMRV, following SIMPLER."
   },
   {
    "text": "Cosmos-Surg-dVRK gave higher success rates than the real surgical robot: mean bias error was 0.140 with human labels and 0.153 with the automated classifier, and 0.325 when the model was trained without failure episodes.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Table 2, Sections 5.2.3-5.2.4",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The authors describe false positives such as a needle snapping into a misaligned gripper, and false negatives such as a needle falling from a correct grasp."
   },
   {
    "text": "Without failure examples in training, the 1X World Model leaned toward predicting success, for example by moving objects into easier positions or misjudging grasp radius.",
    "level": "verified",
    "sources": [
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Section 5.4",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "1X's 2026-01-12 post, where the world model drives the robot as a policy, also says generations can be overly optimistic about task completion and depth."
   },
   {
    "text": "Some world models predict lower success than the real robot: Ctrl-World under-estimated real success rates (fit y = 0.81x - 0.11), and Veo (Robotics) predicted lower absolute success rates than real trials in nominal and out-of-distribution tests while keeping the ranking.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2510.10125v3",
      "title": "Ctrl-World PDF v3: Figure 7",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Sections 3.2, 4.1",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "From Ctrl-World Table 3, the world-model success rate is below the real rate in all 21 policy-task pairs, mean 0.324 vs 0.526 (inferred)."
   },
   {
    "text": "Against the LIBERO simulator as ground truth, the first version of WorldGym (then called WPE) under-estimated an in-distribution policy (for example LIBERO-Spatial 0.91 true vs 0.69 estimated) and over-estimated noise-perturbed out-of-distribution policies.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.00613v1",
      "title": "Evaluating Robot Policies in a World Model (WorldGym arXiv v1): Section 5.1",
      "type": "paper",
      "date": "2025-05-31",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Absolute calibration is unsolved in the Runway study: Runway states it tested ranking only; in its figure the simulated score is above the real RoboArena score for 7 of 8 policies (PaLI-Gemma Binning about 19 vs about 6).",
    "level": "inferred",
    "sources": [
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://d3phaj0sisr2ct.cloudfront.net/research/images/real_vs_gwm1_success_rates-01.png",
      "title": "Runway figure: Real vs. GWM-1 Success Rates",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The ranking-only scope is verified in the post; the 7 of 8 count is read from the figure."
   },
   {
    "text": "Four of the papers name contact-rich interaction as a failure point: Veo (Robotics) shows objects appearing during interaction with small objects; RoboWorld objects disintegrate or change shape after contact; Ctrl-World misses collisions, sliding and rotations; WorldEval shows object deformation, objects appearing or disappearing, overexposure and arm ghosting, mostly for low-performing policies.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 7, Figure 11",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Section 7, Appendix E.3",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.10125",
      "title": "Ctrl-World full text (arXiv HTML v3): Section 5.3",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Appendix A",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "In diverse real scenes, the open-source Ctrl-World model produced heavy hallucinations during object interaction and mis-ranked policies (Pearson r 0.53, MMRV 0.22 in the PolaRiS study).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS full text (arXiv HTML v2): Section 5.2, Figure 7",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Agreement falls for novel objects: Veo (Robotics) policy comparison reached Pearson 0.56 and MMRV 0.15 on the novel-object axis vs 0.91 and 0.0 for background changes. The 1X World Model also struggles with held-out objects.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675v2/ood_policy_comparison.png",
      "title": "Veo report Figure 9",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Section 7",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "VLM judges make errors: in about 100 sampled RoboWorld rollouts GPT-4o scored about one point higher than humans on the 0-5 rubric and sometimes marked unchanged scenes as success; in WorldGym, GPT-4o missed 19% of successful real RT-1 videos (true positive rate 0.81 +/- 0.14, false positive rate 0.03 +/- 0.05).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Appendix E.3",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym full text (arXiv HTML v3): Appendix B.2",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Scoring design changes the result: RoboWorld rank correlation was 0.970 with a progress rubric judged on fixed views, 0.922 with binary success, and 0.862 when the wrist view was used for success judgments.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Section 5.4, Appendix B.4",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Generated surgical videos are uncertain at the edges of small objects (needle, thread, grippers), which makes success labels ambiguous.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Section 6",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Short horizons and human scoring limit current studies: Veo (Robotics) episodes are 8 seconds long and were scored by humans; Runway used about 10 human graders per rollout (over 16,000 ratings for 1,450 rollouts); the 1X World Model accumulates lower-body position error during locomotion.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 7",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Section 7",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Compute cost: RoboWorld needed 100 H100 GPU hours to replicate RoboArena for 8 policies; in RoboWorld's measurement, bidirectional world models (Ctrl-World, PersistWorld) ran at 5.70 frames per second at 4 denoising steps vs 15.31 for autoregressive models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Sections 5.2-5.3",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Pooled correlations can hide weak per-task agreement and many statistics rest on few points: recomputed from WorldGym Table 5, per-task r within a single policy is 0.44-0.52 while the pooled value is 0.78; WorldEval per-task r uses 4 policies; Veo per-axis r uses 5 checkpoints.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym full text (arXiv HTML v3): Table 5",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Table 1",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 4.2",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Computed by us from the published tables and figure point counts."
   },
   {
    "text": "Most agreement numbers are measured by the world model's own builders, and several reuse real-robot numbers from earlier work instead of new paired trials (WorldGym from OpenVLA; RoboWorld and Runway from RoboArena; Cosmos-Surg-dVRK cholecystectomy from Kim et al. 2025). The only third-party measurement found in this sweep, PolaRiS on Ctrl-World, shares an author (Chelsea Finn).",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2512.16881",
      "title": "PolaRiS (arXiv abs, author list)",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World (arXiv abs, author list)",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Derived from the author lists and data-source statements of the nine items in this fragment."
   },
   {
    "text": "World models can predict higher success than real robots achieve. DreamDojo's authors report that its absolute success rates are often higher than real ones; in its Fig. 5a, DreamDojo rates span about 0.06-0.81 while real rates span 0.0-0.44.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.06949",
      "title": "DreamDojo full text (arXiv HTML v1), Sec. 4.7 'Downstream Applications' and Limitations",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949v1/correlation_compressed.svg",
      "title": "DreamDojo Fig. 5(a): real vs DreamDojo success rates (points A-F)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Range values read from the figure.",
    "featured": 6
   },
   {
    "text": "Both Ctrl-World and its RL-post-trained version PersistWorld make tasks look easier than they are: world-model task progress spans about 0.79-1.0 while real progress spans about 0.33-1.0 for the same 9 task-policy pairs.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Range values read from Fig. 8."
   },
   {
    "text": "Video generators used as evaluators show an optimistic bias on contact-sensitive failures. In WMBench, one CVPR 2026 challenge world model overestimated success on all 12 subtasks (by +0.04 to +0.27) and another on 11 of 12 (up to +0.30).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/success_rate_difference_bar.svg",
      "title": "GigaWorld-1 Fig. 17: Gen - Real success-rate difference per subtask and model",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Under-prediction also occurs. Veo (Robotics) predicted success rates of about 0.0-0.3 for policies whose real success was about 0.05-0.69; WEAVER's authors report that pretrained world models tend to underestimate policy performance.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo (Robotics) report full text (arXiv HTML v2), Sec. 3-4 and 7",
      "type": "paper",
      "date": "2026-01-06",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/nominal_correlation.png",
      "title": "Veo (Robotics) Fig. 4: nominal real vs predicted success (MMRV, Pearson)",
      "type": "paper",
      "date": "2026-01-06",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Veo ranges read from Fig. 4."
   },
   {
    "text": "World models trained on success-biased human demonstrations tend to produce 'hallucinated success'. In PlayWorld, models trained on human data reached Pearson r of 0.6636 and 0.6619 with real success, against 0.8766 for the model trained on autonomous play data.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.09030",
      "title": "PlayWorld full text (arXiv HTML v3), Sec. 4.4 and Fig. 7",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v3/images/experiment/correlation_new.png",
      "title": "PlayWorld Fig. 7: policy evaluation success-rate correlation (RMSE and r per training-data source)",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Training data without failures weakens evaluation. In NVIDIA's RoboMimic study, removing policy rollouts from world-model training lowered Pearson correlation from 0.847 to 0.505 on Square and from 0.833 to 0.502 on Tool Hang.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": ""
   },
   {
    "text": "Correct ranking can coexist with wrong absolute success rates. Pelican-Sim reaches Spearman 1.0 and MMRV 0 with 200 adaptation rollouts while its success-rate error is still 4.0 percentage points; with no adaptation the error is 8.6 points.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Errors compound over long autoregressive rollouts. PersistWorld's authors report manipulated objects losing their identity within seconds; in WMBench, SVD's multi-view PSNR falls from 14.05 (0-8 s) to 6.88 (32-40 s).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Appearance-stability metrics can reward static, action-ignoring videos. In WMBench, Background Consistency (rho = -0.45) and Photometric Consistency (rho = -0.42) correlate negatively with human-judged evaluator quality (WMES).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Automatic VLM judges are imperfect. OSCAR's GPT-5 judge matched human labels on 78 of 100 real clips (recall 0.66); NVIDIA reports VLM annotation accuracy of 65-80%; WMBench's VLM-judged Interaction Quality correlates at rho = -0.11 with WMES.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.04463",
      "title": "OSCAR full text (arXiv HTML v2), Sec. 5.4 Table 4, Appendix A.10",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Contact-rich, granular and deformable interactions remain hard to simulate. WEAVER names bean pouring and bag and towel handling as the hardest tasks; Veo (Robotics) reports objects appearing spontaneously during contact and its weakest out-of-distribution correlation on novel objects (Pearson 0.56).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo (Robotics) report full text (arXiv HTML v2), Sec. 3-4 and 7",
      "type": "paper",
      "date": "2026-01-06",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/ood_policy_comparison.png",
      "title": "Veo (Robotics) Fig. 9: OOD axes, real vs predicted success (MMRV, Pearson)",
      "type": "paper",
      "date": "2026-01-06",
      "accessed": "2026-10-11"
     }
    ],
    "note": "",
    "featured": 8
   },
   {
    "text": "Ranking errors persist on hard tasks: in DexTouch-WM both adapted world models reverse the order of pi0.5 and X-VLA on Stand Bottle (MMRV 0.0667; Pearson 0.334 and 0.577).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.20649",
      "title": "DexTouch-WM full text (arXiv HTML v2), Sec. IV-D, Tables II-III",
      "type": "paper",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Many agreement statistics rest on very few points: DexTouch-WM uses 3 policies per task, Pelican-Sim 5 checkpoints, DreamDojo 6 checkpoints, PersistWorld 9 task-policy pairs, and WEAVER 10 points.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.20649",
      "title": "DexTouch-WM full text (arXiv HTML v2), Sec. IV-D, Tables II-III",
      "type": "paper",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949v1/correlation_compressed.svg",
      "title": "DreamDojo Fig. 5(a): real vs DreamDojo success rates (points A-F)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Our reading across papers."
   },
   {
    "text": "Several protocols replay fixed action sequences instead of letting the policy react to generated observations. Pelican-Sim states its matched-action protocol evaluates outcome prediction for fixed trajectories, not closed-loop execution; WEAVER and OSCAR replay recorded real actions.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.04463",
      "title": "OSCAR full text (arXiv HTML v2), Sec. 5.4 Table 4, Appendix A.10",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     }
    ],
    "note": ""
   },
   {
    "text": "Evaluators often need task-specific fine-tuning. WEAVER's Pearson correlation with real success rises from 0.563 (pretrained) to 0.863 after fine-tuning on 50 rollouts per task; Pelican-Sim's Pearson rises from 0.902 with no adaptation to 0.994 with 1,000 RoboTwin rollouts.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2609.12036",
      "title": "Pelican-Sim 1.0 full text (arXiv HTML v1), Sec. 4.6, Tables 10-11",
      "type": "paper",
      "date": "2026-09-10",
      "accessed": "2026-10-10"
     }
    ],
    "note": ""
   },
   {
    "text": "Generated rollouts are short in some systems: Veo (Robotics) policy rollouts are 8-second episodes, and its authors list minute-long multi-view generation as an open milestone.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo (Robotics) report full text (arXiv HTML v2), Sec. 3-4 and 7",
      "type": "paper",
      "date": "2026-01-06",
      "accessed": "2026-10-11"
     }
    ],
    "note": ""
   },
   {
    "text": "Public checkpoints may differ from the evaluated model: EnerVerse-AC's released weights were trained only on open-source AgiBot World data without the failure trajectories used in the paper, because of commercial restrictions.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/AgibotTech/EnerVerse-AC",
      "title": "AgibotTech/EnerVerse-AC repository README (License section; no LICENSE file)",
      "type": "repo",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": ""
   },
   {
    "text": "WorldArena reports a perception-functionality gap: its video-quality index EWMScore tracks human ratings (Pearson r = 0.825, 14 models) but tracks downstream usefulness much less, with r = 0.600 against data-engine performance and r = 0.360 against action-planner performance (6 models each).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/ewm_human_dataengine_actionplanner_3subplots_14models.svg",
      "title": "WorldArena Figure 5: EWMScore vs human evaluation, data engine and action planner",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "As action planners, the best world models reached 20% and 35% on the two tasks versus 77% and 66% for the pi0.5 policy (Table 5)."
   },
   {
    "text": "In WorldArena's policy-evaluator test, both action-conditioned world models (Ctrl-World and Cosmos-Predict 2.5) gave higher success rates than the RoboTwin simulator for the same policies, and Cosmos-Predict 2.5 ranked the 5 policies poorly (Pearson r = 0.483 versus 0.986 for Ctrl-World).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.08971",
      "title": "WorldArena full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.08971v2/simulator_vs_worldmodel_correlation.svg",
      "title": "WorldArena Figure 4: correlation of policy evaluation results from world models and the physical simulator",
      "type": "paper",
      "date": "2026-02-11",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Authors attribute the upward bias to partial overfitting to successful trajectories."
   },
   {
    "text": "WorldArena 2.0 finds that simulator results do not predict real-robot results for world models: task-success rankings of 6 world models correlate between the two simulators (Spearman rho = 0.771) but only weakly with real-robot success (rho = 0.348 for RoboTwin vs real, 0.522 for LIBERO vs real; neither significant).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2605.17912",
      "title": "WorldArena 2.0 full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2605.17912v1/task_success_correlation.svg",
      "title": "WorldArena 2.0 Figure 6: cross-platform task success rate correlation",
      "type": "paper",
      "date": "2026-05-18",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Authors conclude simulation performance, perceptual or functional, is not a reliable proxy for real-world deployment."
   },
   {
    "text": "World-in-World finds that visual quality does not predict closed-loop task success: on its leaderboard of 16 Active Recognition entries, zero-shot Cosmos-Predict2 has the highest generation-quality score (0.481732) but the lowest success rate (55.35%, tied with Wan2.2 5B), while controllability (1 - LPIPS) tracks success more closely.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.18135v2",
      "title": "World-in-World full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-08-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.18135v2/merged_final_scatter_plot.svg",
      "title": "World-in-World Figure 5: SR vs generation quality and SR vs controllability in AR",
      "type": "paper",
      "date": "2026-08-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-in-world.github.io/subpages/leaderboard.html",
      "title": "World-In-World Leaderboard",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The paper reports no correlation coefficient. Computed by us from the 16 leaderboard entries (inferred): Pearson r = 0.119 and Spearman rho = -0.007 between generation quality and AR success. Gains in manipulation were small: best post-trained model SVD 46.5% success versus 44.5% for the VLM policy without a world model.",
    "featured": 4
   },
   {
    "text": "DreamGen Bench's authors report that their physics judge VideoCon-Physics was not trained on multi-view (RoboCasa) or diverse robot videos, so they average it with a general VLM, and that the lightweight automatic judges can hallucinate when rating physical realism.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.12705v2",
      "title": "DreamGen full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2025-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://raw.githubusercontent.com/NVIDIA/GR00T-Dreams/main/README.md",
      "title": "NVIDIA/GR00T-Dreams README, section 5 DreamGen Bench",
      "type": "repo",
      "date": null,
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "WorldSimBench finds that GPT-4o as a judge of embodied video quality agrees poorly with human ratings: Pearson correlation 0.28 for driving and 0.07 for manipulation videos, and -0.04 and -0.06 in two held-out-generator tests.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2410.18072",
      "title": "WorldSimBench full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2024-10-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/qin25f/qin25f.pdf",
      "title": "WorldSimBench PMLR PDF (affiliation footnote, Table 3)",
      "type": "paper",
      "date": "2025-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "In WoW-World-Eval's IDM Turing test, actions recovered from most models' generated videos rarely succeed on a real robot: CogVideoX, Cosmos-Predict1 and Wan2.1 reach 0.00%, Hailuo 2.47% and Kling 9.88%, while the authors' WoW-wan reaches 40.74%. Hailuo has the highest benchmark score (52.55) yet 2.47% execution success.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The IDM succeeds 90% of the time on ground-truth videos, so the authors attribute failures to the generated videos."
   },
   {
    "text": "WoW-World-Eval's automatic planning score agrees only moderately with human planning ratings (Pearson r = 0.43, Spearman rho = 0.51), lower than its other dimensions (0.66 to 0.81).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.04137",
      "title": "WoW-World-Eval full text (arXiv HTML v1)",
      "type": "paper",
      "date": "2026-01-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "RoboWM-Bench finds that a perceptual benchmark saturates on manipulation videos: PAI-Bench average quality scores cluster around 0.78 across models and tasks, while execution success of the same videos varies widely (robot task-level success from 0% to 90%).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.19092v2",
      "title": "RoboWM-Bench full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-05-14",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Appendix D and Figure 6 of the paper. Generated robot arms also show geometric distortions, so the recovered joint configuration can miss the target even when the video looks correct."
   },
   {
    "text": "The Open-H-Embodiment paper notes that pixel metrics for its surgical world model (L1, SSIM) need recorded actions and ground-truth video, so they cannot score closed-loop rollouts driven by a policy, and they do not capture whether instruments are in the right place or tool-tissue contact is plausible; early tool-segmentation metrics were not reliable across embodiments.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.21017v3",
      "title": "Open-H-Embodiment full text (arXiv HTML v3), Cosmos-H-Surgical-Simulator sections",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Visual realism does not track physical correctness in Physics-IQ. A multimodal LLM (Gemini 1.5 Pro) asked to spot the generated video in a real/generated pair found Sora hardest to detect (55.6%, chance 50%), yet Sora had the lowest Physics-IQ score (10.0%); across 8 model variants the correlation between realism and Physics-IQ score was Pearson r = -0.46, p = .249, not significant. The best model scored 29.5% of the real-video ceiling.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2501.09038",
      "title": "Do generative video models understand physical principles? (arXiv HTML v3), Figure 5",
      "type": "paper",
      "date": "2025-02-27",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The MLLM's stated reasons for its choices were often unrelated to the video content. The authors also state that PSNR, SSIM, FVD and LPIPS are not designed to judge physics."
   },
   {
    "text": "An audit of Physics-IQ found measurement errors in its prompts and ground truth: of the 198 benchmark videos, 69 had unclear prompts and 59 had artifacts (motion not caused by the tested effect), 20 had both. The fixes refine 57.6% of samples and over 34.8% of prompts, and change the ranking of six image-to-video models (Kendall tau = 0.46, Spearman rho = 0.65). Artifact removal lowered all IoU-based scores significantly (Wilcoxon p << 1e-5, Cohen's d <= -1).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1), Figure 3 and Section 4",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Unclear-prompt types in order of severity: factually incorrect, temporally imprecise, omitted key information, vague language. Wan 2.2 dropped from first to third after artifact removal; the authors say part of its lead came from confounding effects."
   },
   {
    "text": "The original Physics-IQ score normalises by physical variation averaged over the whole dataset, so experiments with low trial-to-trial variation can never reach full score and those with high variation can exceed it, giving samples unequal weight; the score also cannot be traced to individual samples.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1), Section 3.2 and Appendix C.6",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Physics-IQ Verified replaces it with a per-sample score; in its tests the new formula raised all scores but did not change the ranking. Under the verified protocol the spatiotemporal metric's physical variance rises by about 17%, so scores should not be compared across protocols."
   },
   {
    "text": "Physics scores depend on prompt wording. In Physics-IQ Verified, Sora 2 scored 16.7 ± 0.8 with the original prompts and 27.3 ± 0.8 with model-specific best-practice prompts, while Wan 2.2 scored lower with best-practice prompts. In PhyGenBench, rewriting prompts with the expected outcome raised Kling from 0.49 to 0.56 and CogVideoX-5B from 0.45 to 0.52. VBench-2.0 reports that models using an external prompt refiner did reasonably well on Physics, which the authors read as prompting partly compensating for missing physical reasoning.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1), Table 6 and Section 4.2",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2410.05363",
      "title": "PhyGenBench paper (arXiv HTML v1), Table 11",
      "type": "paper",
      "date": "2024-10-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.21755",
      "title": "VBench-2.0 (arXiv HTML v2), Section V-C",
      "type": "paper",
      "date": "2025-08-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The Sora 2 values are Physics-IQ Verified scores with the original (uncleaned) ground truth; with cleaned ground truth they are 15.7 ± 0.7 and 26.5 ± 0.8."
   },
   {
    "text": "Results for closed, API-only models drift over time. A single Sora 2 run from October 2025 scored 42.8 on the original Physics-IQ score (original prompts and ground truth), while runs in April 2026 scored 12.7 ± 0.8 under the same setting.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.18943",
      "title": "Physics-IQ Verified (arXiv HTML v1), Appendix E.2, Table 5",
      "type": "paper",
      "date": "2026-06-17",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The October 2025 value is a single run without a standard deviation. The Physics-IQ original leaderboard lists Sora2 at 42.3%, reported in arXiv 2601.10553."
   },
   {
    "text": "Leaderboard scores mix single generations with best-of-N selection by reward models, and the test data can leak into selection. On the Physics-IQ Verified leaderboard, Odyssey-3 Pro scores 63.37 ± 0.63 with single generations and 66.10 with best-of-8 selection; the README forbids using test data for best-of-N selection or prompt rewriting and reports that frontier models downloaded the test data on their own when asked to optimise prompts.",
    "level": "verified",
    "sources": [
     {
      "url": "https://github.com/google-deepmind/physics-IQ-benchmark#leaderboard",
      "title": "Physics-IQ Verified leaderboard and rules (repo README)",
      "type": "leaderboard",
      "date": "2026-10-08",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Both Odyssey entries use custom prompts and were added on 2026-10-08. On the original Physics-IQ leaderboard the top three video-to-video entries are also best-of-N runs (GeoPhys or WMReward)."
   },
   {
    "text": "Several world-model leaderboards rank self-reported results next to organiser-run results. On the WorldScore leaderboard (35 rows), the top entries for both WorldScore-Static (UniWorld-View, 85.53) and WorldScore-Dynamic (WorldScape-0.2(MoE), 76.23) were sampled and evaluated by their own teams, while the best rows evaluated by the WorldScore team score 72.69 (static) and 59.12 (dynamic). The VBench image-to-video table is topped by a self-evaluated entry (DreamX-World-1.0, 90.49%).",
    "level": "verified",
    "sources": [
     {
      "url": "https://huggingface.co/spaces/Howieeeee/WorldScore_Leaderboard",
      "title": "WorldScore Leaderboard (leaderboard.csv)",
      "type": "leaderboard",
      "date": "2026-09-10",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://huggingface.co/spaces/Vchitect/VBench_Leaderboard",
      "title": "VBench Leaderboard",
      "type": "leaderboard",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The 'Evaluated by' columns are shown on both leaderboards; whether self-run numbers differ systematically from organiser-run numbers has not been measured in the sources read."
   },
   {
    "text": "General video-quality scores are weak or unrelated predictors of physics judgements. In VideoPhy, frame aesthetics (LAION) correlates 0.3 with human physical-commonsense scores and motion magnitude -0.8; Gen-2 had one of the highest aesthetics scores (5.8) but only 7.6% of its videos passed both caption and physics checks. In VideoPhy-2, physical commonsense correlates 0.09 with aesthetics and 0.002 with motion.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2406.03520",
      "title": "VideoPhy (arXiv HTML v2), Appendix O, Table 11",
      "type": "paper",
      "date": "2024-10-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.06800",
      "title": "VideoPhy-2 (arXiv HTML v1), Table 3",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Pearson correlations; the unit of analysis (model or video) is not stated in the VideoPhy appendix."
   },
   {
    "text": "VBench does not separate models by physics adherence. WorldModelBench compared pairwise win rates of 8 models: frame-wise quality win rates from the two benchmarks correlate 0.69, but WorldModelBench physics-adherence win rates correlate only 0.28 with VBench overall win rates. Within WorldModelBench, Luma has higher frame-wise (0.81 vs 0.63) and temporal quality (0.76 vs 0.63) than Mochi but completes fewer tasks (44% vs 53%) with similar physics adherence (4.13 vs 4.14).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2502.20694",
      "title": "WorldModelBench (arXiv HTML v1), Section 4.1 and 4.3, Figure 9",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Models in the comparison: Opensora, Pandora, Luma, Minimax, Mochi, CogVideoX, Kling, Runway."
   },
   {
    "text": "Raising visual quality does not raise physics scores. In PhyGenBench, enhancing Vchitect 2.0 videos with VEnhancer lifted it on VBench above Kling, but its PhyGenBench physical-commonsense average stayed at 0.45 (Spearman 0.86 between scores before and after). Existing automatic evaluators correlated poorly with human physics ratings on 512 videos: VideoScore Spearman 0.19, DEVIL 0.18, VideoPhy's evaluator 0.04. Scaling CogVideoX from 2B to 5B raised the average from 0.39 to 0.45 but mechanics only from 0.38 to 0.39.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2410.05363",
      "title": "PhyGenBench paper (arXiv HTML v1), Tables 1, 2 and 12, Appendix D",
      "type": "paper",
      "date": "2024-10-07",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The VEnhancer result covers one model."
   },
   {
    "text": "General-purpose VLM judges are close to chance on physics. On VideoPhy, GPT-4-Vision reached ROC-AUC 53 for physical commonsense and Gemini-1.5-Pro 58 (54 in the text); on VideoPhy-2, Gemini-2.0-Flash-Exp reached Pearson r = 0.11 with human physics scores; on PhyWorldBench, plain GPT-o1 reached ROC-AUC 61.6 for physics. PhyWorldBench reports that MLLM judges tend to rationalise unrealistic content unless told the video is generated, and show an 'aesthetic bias' toward visually appealing videos.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2406.03520",
      "title": "VideoPhy (arXiv HTML v2), Table 4",
      "type": "paper",
      "date": "2024-10-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.06800",
      "title": "VideoPhy-2 (arXiv HTML v1), Table 4",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2507.13428",
      "title": "PhyWorldBench (arXiv HTML), Section 3.3 and Appendix D",
      "type": "paper",
      "date": "2026-05-26",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Fine-tuned or prompted judges do better (VideoCon-Physics 73, CAP 75.1 ROC-AUC) but remain well below human agreement levels."
   },
   {
    "text": "Fine-tuned physics judges can fail on real video. VideoPhy-2-AutoEval gave clean real-world recordings of Newtonian experiments a mean physical-commonsense score of 3.71 of 5, barely above generated videos (3.53), passed only 23.6% of real videos on its joint physics check (13.9% for generated) and passed 0% of real projectiles, bouncing balls, sliding books, collisions and rolling cans.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2504.02918",
      "title": "Morpheus (arXiv HTML v3), Section 4 and Appendix D.6",
      "type": "paper",
      "date": "2026-06-29",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Independent test by the Morpheus authors (no overlap with VideoPhy-2 authors). The same paper finds human ratings correlate with motion plausibility (rho = 0.768) but not with conservation-law checks (rho = -0.157), so human raters also miss quantitative violations."
   },
   {
    "text": "Human raters disagree often on physics, so human labels used to validate judges are noisy. Inter-annotator agreement was 70% for physical commonsense and 75% for semantic adherence in VideoPhy, 75% to 80% in VideoPhy-2, and 70.0% pairwise agreement in WorldModelBench; in WorldPrediction it was 0.73 (world modeling) and 0.65 (procedural planning) before filtering.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2406.03520",
      "title": "VideoPhy (arXiv HTML v2), Section 4",
      "type": "paper",
      "date": "2024-10-03",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.06800",
      "title": "VideoPhy-2 (arXiv HTML v1), Section 4",
      "type": "paper",
      "date": "2025-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2502.20694",
      "title": "WorldModelBench (arXiv HTML v1), Table 2",
      "type": "paper",
      "date": "2025-02-28",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2506.04363",
      "title": "WorldPrediction (arXiv HTML v1), Appendix A.1, Table 4",
      "type": "paper",
      "date": "2025-06-04",
      "accessed": "2026-10-10"
     }
    ],
    "note": "WorldPrediction then kept only samples that both annotators answered correctly (825 of 1,500 for WM, 570 of 1,500 for PP), so its reported human accuracy is perfect by construction."
   },
   {
    "text": "Human-alignment checks for VBench and VBench-2.0 rest on four models per dimension. Each reported per-dimension correlation (60.73% to 99.80% for VBench; 81.70% to 99.46% for VBench-2.0) compares four model-level win ratios, so a single model can swing the value.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2311.17982",
      "title": "VBench (arXiv HTML v1), Table A5",
      "type": "paper",
      "date": "2023-11-29",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2503.21755",
      "title": "VBench-2.0 (arXiv HTML v2), Table IV",
      "type": "paper",
      "date": "2025-08-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Inferred from the tables, which list four models each. The figures label the statistic Spearman's rho, but values like 96.51% are not possible for a rank correlation over four points."
   },
   {
    "text": "Fréchet Video Distance (FVD) favours per-frame quality over motion. Videos with mild spatial distortion and no temporal corruption got FVD 317.10, while videos with slightly less spatial distortion but severe temporal inconsistency got a better FVD of 310.52; temporal inconsistency raised FVD by only 3% on FaceForensics. Resampling motionless (frozen) DIGAN videos cut FVD on UCF-101 from 1303.13 to 715.96 (-45.1%) without any motion. Features from the self-supervised VideoMAE-v2 make FVD about five times more sensitive to temporal corruption than the standard I3D features.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2404.12391",
      "title": "On the Content Bias in Fréchet Video Distance (arXiv abstract page; CVPR 2024)",
      "type": "paper",
      "date": "2024-04-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2404.12391",
      "title": "On the Content Bias in Fréchet Video Distance (arXiv HTML), Figure 1, Section 3, Table 2",
      "type": "paper",
      "date": "2024-04-18",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Authors: University of Maryland, Carnegie Mellon University, Adobe Research. They attribute the bias to I3D features trained on the content-biased Kinetics dataset. Venue CVPR 2024 per arXiv comments."
   },
   {
    "text": "Video generators trained on simulated physics generalise within the training distribution but fail outside it, and scaling does not fix this. For a DiT-L model on uniform motion with 3M training videos, the velocity error was 0.012 in distribution and 0.427 out of distribution; out-of-distribution errors were an order of magnitude higher in all settings and did not fall with more data or larger models (DiT-B: 0.433, 0.328, 0.358 at 30K, 300K, 3M videos). Models imitated the closest training example ('case-based' generalisation), prioritising color > size > velocity > shape. Scaling data from 600K to 6M cut abnormal videos in combinatorial scenes from 67% to 10%.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/abs/2411.02385",
      "title": "How Far is Video Generation from World Model: A Physical Law Perspective (arXiv abstract page; ICML 2025)",
      "type": "paper",
      "date": "2024-11-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2411.02385",
      "title": "How Far is Video Generation from World Model (arXiv HTML v2), Sections 1, 3 and 4",
      "type": "paper",
      "date": "2025-06-22",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Authors: ByteDance Research and Tsinghua University. Uses a 2D Box2D testbed (uniform motion, elastic collision, parabolic motion). Venue ICML 2025 per arXiv comments. The 67% to 10% abnormal rates come from manual labelling."
   },
   {
    "text": "VBench-2.0's authors state that judging video by visual quality is a common bias: CogVideoX scores relatively well on many VBench-2.0 intrinsic-faithfulness dimensions despite a lower VBench quality score than Sora and HunyuanVideo, while HunyuanVideo has high visual quality but trails on structure-driven dimensions. They also report that models fail simple dynamic attribute and position changes about 80% of the time.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2503.21755",
      "title": "VBench-2.0 (arXiv HTML v2), Sections V-B and V-D",
      "type": "paper",
      "date": "2025-08-20",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Based on four models."
   },
   {
    "text": "In WorldLens, a pretrained end-to-end planner driving inside generated worlds completed only 6.89% to 13.51% of routes in closed loop (Arena Driving Score 4.82% to 10.59%, 5 world models), while the same models scored 71.23% to 78.91% PDMS in open loop.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2512.10958v2",
      "title": "WorldLens arXiv PDF v2, Table 2 and Tables 17-19",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Closed-loop tests use two nuScenes maps aligned with DriveArena and five simulation sequences. The paper names UniAD or VAD as the planner. Failures are collisions and off-road drift. The result mixes planner weakness with world-model weakness; the paper reads it as world models being inadequate substitutes for real data in control."
   },
   {
    "text": "Human raters in WorldLens gave current driving world models mean scores between 2.036 and 2.961 on a 1 to 10 scale across six models and six dimensions (realism, vehicle and pedestrian realism, physical plausibility, 3D & 4D consistency, behavioral safety); the paper summarizes this as 2 to 3 out of 10.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML v2, Section 5.2 and Tables 24-29",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "The lowest score given was 2.0 for every model and dimension, so the scale was used mainly at the low end."
   },
   {
    "text": "Perception models trained on real data lose much of their accuracy on generated driving video: in WorldLens, BEVFusion 3D detection NDS on six models' videos was 21.96% to 33.22% against 44.72% on the real reference, and tracking AMOTA 6.90% to 15.30% against 36.30%. In DriveArena, UniAD detection mAP fell from 37.98 on real nuScenes images to 16.06 on generated images.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML v2, Table 3",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/pdf/2408.00415v1",
      "title": "DriveArena arXiv PDF v1, Table 1",
      "type": "paper",
      "date": "2024-08-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "WorldLens's 'Empirical Max' column is used here as the real-data reference."
   },
   {
    "text": "Visual-quality metrics do not predict whether a generated world is usable. In WorldLens, OpenDWM had the best subject fidelity and was rated highest on vehicle realism, yet scored about 30% lower than DiST-4D on 3D detection (NDS 21.96% vs 33.22%); DiST-4D had the lowest FVD but lower subject fidelity than OpenDWM. ACT-Bench's project page notes that Vista, which it calls the leading model on FID/FVD, followed instructed actions in only 30.72% of videos.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML v2, Sections 5.1 and 5.3",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://turingmotors.github.io/actbench/",
      "title": "ACT-Bench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "ACT-Bench's paper does not report FID/FVD values for Vista or Terra."
   },
   {
    "text": "Driving world models often do not perform the commanded action. In ACT-Bench, the generated motion matched the instructed high-level action in 30.72% of 2,286 videos for Vista and 44.11% for Terra (63.2% for Terra v2 on the project page). The paper also reports 'causal misalignment', where an instruction to the ego car changes other cars' behaviour (a lead car stops when the ego car is told to stop).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2412.05337",
      "title": "ACT-Bench arXiv HTML v1, Sections 5.1 and 5.3",
      "type": "paper",
      "date": "2024-12-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://turingmotors.github.io/actbench/",
      "title": "ACT-Bench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     }
    ],
    "note": "The causal-misalignment finding is shown with examples; no frequency is given."
   },
   {
    "text": "Close agreement between scores on real and generated images can reflect a planner that ignores the images. In DriveArena, UniAD's open-loop PDMS was 0.910 on real nuScenes images and 0.902 on generated images; the authors attribute the small gap partly to UniAD's heavy reliance on ego-state inputs. On DriveArena's own simulated routes UniAD's PDMS was 0.636, and in closed loop it completed 13.7% of route length on average over 4 routes (ADS 0.086).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2408.00415v1",
      "title": "DriveArena arXiv PDF v1, Tables 2 and 3",
      "type": "paper",
      "date": "2024-08-01",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Replaying logged camera images fails once the simulated car leaves the recorded path, and generative closed-loop testing is limited by available sensor data. In Bench2Drive-R's nuPlan closed-loop test, detection NDS on log-replayed images was 0.05 vs 28.23 on generated images; only 10% of nuPlan scenes have sensor data, so the test used 10 clips from each of 14 scenario types and a single planner (VAD), whose R-CLS rose only from 27.24 (log replay) to 30.49.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2412.09647v1",
      "title": "Bench2Drive-R arXiv PDF v1, Section 4.2.1 and Table 6",
      "type": "paper",
      "date": "2024-12-11",
      "accessed": "2026-10-10"
     }
    ]
   },
   {
    "text": "Trajectory metrics recovered from generated video agree less with human judgement than video metrics. In DrivingGen, Spearman rho with human win rates was 0.6261 to 0.7682 for trajectory realism, quality and consistency, against 0.8344 to 0.8899 for the video metrics; the authors attribute this to noisy monocular SLAM on videos with artifacts. Under ego-trajectory conditioning, models showed large ADE/DTW errors, partly because artifacts impair trajectory recovery.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML v2, Section 4.1 and Figure 5",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Common driving evaluation sets are dominated by easy conditions: DrivingGen reports that over 80% of the nuScenes validation data and 90% of the OpenDV validation data were collected in normal sunny daytime conditions.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML v2, Section 3.1",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Consistency and smoothness metrics reward videos with little motion. DrivingGen states that fixed-stride frame-consistency metrics can be gamed by near-static videos and adds motion-adaptive sampling. PlayWorld reports that Depth Stability and Subject Consistency stay high when a model produces little or no camera motion, even when the objective is not completed; motion smoothness was between 97.84% and 99.46% for all nine PlayWorld models.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML v2, Section 3.2.3",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2608.13552v2",
      "title": "PlayWorld arXiv HTML v2, Section 4.3 and Table 6",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Some world-model metrics no longer separate models. In GameWorld Score, motion smoothness was 0.98 for all three models and temporal consistency 0.94 to 0.97. In WorldMark v2, Local Memory had eight of ten models within 8 points of the best (standard deviation 2.6), while Global Memory ranged from 34.0 to 75.8 (standard deviation 13.1).",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.18701",
      "title": "Matrix-Game arXiv HTML v1, Table 2",
      "type": "paper",
      "date": "2025-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.21686v2",
      "title": "WorldMark arXiv HTML v2, Section 4.2",
      "type": "paper",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     }
    ],
    "note": "Inferred for GameWorld Score from its Table 2 values; WorldMark states the Local Memory saturation itself."
   },
   {
    "text": "Ranking interactive world models by appearance can misrank them as interactive environments. In WorldMark v2, Perceptual Quality correlated negatively with translational action dynamics across ten models, with the largest negative value for Response Latency (rho = -0.82); Yume 1.5 ranked first on Perceptual Quality (87.2) and last on translational Direction Accuracy (51.9) and Latency (40.7). The fastest responders were often the least stable (e.g. DreamX-World: latency score 96.1/97.0, rotational stability 27.7).",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.21686v2",
      "title": "WorldMark arXiv HTML v2, Section 4.2",
      "type": "paper",
      "date": "2026-08-05",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Controllability is a separate capability from visual quality, memory and physics. Across 20 models in WBench, navigation scores had near-zero correlation with video quality (r = -0.12), consistency (r = -0.05) and physics compliance (r = -0.15), while physics scores correlated with video quality (r = 0.84). Navigation scores dropped by 33 points from the first turn to turns 4 and later.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2605.25874v1",
      "title": "WBench arXiv HTML v1, Section 5.3 and Figures 3-4",
      "type": "paper",
      "date": "2026-05-25",
      "accessed": "2026-10-11"
     }
    ],
    "note": "The physics-vs-video-quality correlation suggests the physics metric may partly track rendering quality; the authors read it as physics plausibility coming from generative priors."
   },
   {
    "text": "Short-horizon metrics do not predict long-horizon world-model ability. In PlayWorld, HappyOyster had the highest Basic Ability Score (76.4 vs 72.2 for Genie 3) but Genie 3 ranked first on the long-horizon rubric (overall 2.12 vs 1.92 on a 1 to 5 scale); SANA-WM reached the objective region in 80.4% of rollouts (second highest) but its rubric overall was 1.48. No model scored above 1.95 on either state-evolution dimension.",
    "level": "verified",
    "sources": [
     {
      "url": "https://arxiv.org/html/2608.13552v2",
      "title": "PlayWorld arXiv HTML v2, Tables 2, 3 and 6, Section 4.3",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     }
    ]
   },
   {
    "text": "Driving companies describe world models as tools for evaluating driving policies but publish no agreement statistic. Waymo's World Model post (2026-02-06) describes simulation of rare events with no evaluation numbers. Wayve's GAIA-3 page (2025-12-02) says correlation studies against on-road experiments indicate the model can 'reliably predict relative policy performance', with no coefficient, sample size or method. NVIDIA's Alpamayo-R1 paper evaluates in AlpaSim, a 3D Gaussian Splatting reconstruction simulator, and reports no sim-to-real agreement figure.",
    "level": "verified",
    "sources": [
     {
      "url": "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation",
      "title": "The Waymo World Model: A New Frontier For Autonomous Driving Simulation",
      "type": "blog",
      "date": "2026-02-06",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://wayve.ai/thinking/gaia-3/",
      "title": "GAIA-3: Scaling World Models to Power Safety and Evaluation",
      "type": "blog",
      "date": "2025-12-02",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2511.00088",
      "title": "Alpamayo-R1: Bridging Reasoning and Action Prediction for Generalizable Autonomous Driving in the Long Tail",
      "type": "paper",
      "date": "2025-10-30",
      "accessed": "2026-10-11"
     }
    ],
    "note": "This is an absence finding for the pages listed, checked on 2026-10-11. The Wayve page returned HTTP 403 to curl and was read through WebFetch (two passes, including a verbatim extraction request). Tesla was not checked."
   },
   {
    "text": "Human-agreement checks for world-model benchmarks are run by the benchmark's own authors and often on few models: GameWorld Score compared 3 models, WorldMark v1 6, WBench 4 to 6 per aspect, PlayWorld 9, DrivingGen 14; WorldLens reported no number for its evaluator agent. None of the eight benchmarks checked here reports an independent replication.",
    "level": "inferred",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.18701",
      "title": "Matrix-Game arXiv HTML v1",
      "type": "paper",
      "date": "2025-06-23",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.21686v1",
      "title": "WorldMark arXiv HTML v1",
      "type": "paper",
      "date": "2026-04-23",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2605.25874v1",
      "title": "WBench arXiv HTML v1",
      "type": "paper",
      "date": "2026-05-25",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2608.13552v2",
      "title": "PlayWorld arXiv HTML v2",
      "type": "paper",
      "date": "2026-08-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2601.01528v2",
      "title": "DrivingGen arXiv HTML v2",
      "type": "paper",
      "date": "2026-03-07",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2512.10958v2",
      "title": "WorldLens arXiv HTML v2",
      "type": "paper",
      "date": "2026-06-01",
      "accessed": "2026-10-10"
     }
    ],
    "note": "Compiled from the validity_studies rows in this file; 'independent replication' was not searched for beyond these papers."
   }
  ],
  "agreement": [
   {
    "wm": "WorldEval",
    "cat": "own",
    "r": 0.942,
    "n": "4 policies, 3 tasks (average)",
    "idx": 2,
    "study": "WorldEval vs real-robot success rates",
    "statistic": "Table 1 (3 tasks): Pearson r = 0.958 (Place Cup), 0.887 (Strike Block), 0.980 (Handover Block), average 0.942; MMRV = 0.000, 0.133, 0.000, average 0.044. Figure 4 (5 tasks): Bussing Table r = 0.935, MMRV = 0.000; Collect Toy r = 0.885, MMRV = 0.133; Place Cup r = 0.958, MMRV = 0.000; Handover Block r = 0.980, MMRV = 0.000; Strike Block r = 0.887, MMRV = 0.000.",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.19017",
      "title": "WorldEval full text (arXiv HTML v1): Table 1, Section 4",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.19017v1/muti_task_corr.png",
      "title": "WorldEval Figure 4: real vs WorldEval success rates per task",
      "type": "paper",
      "date": "2025-05-25",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "WorldGym",
    "cat": "own",
    "r": 0.78,
    "n": "3 policies",
    "idx": 7,
    "study": "WorldGym vs OpenVLA Bridge real-robot results",
    "statistic": "Pearson r = 0.78 between per-task success rates in WorldGym and in the real world (each point is one task-policy pair). Mean success rate, real vs WorldGym: RT-1-X 18.5% vs 15.5%, Octo 20.0% vs 23.82%, OpenVLA 70.6% vs 67.4%; average difference 3.3%. The order of the three policies by mean success rate is the same in both.",
    "sources": [
     {
      "url": "https://arxiv.org/html/2506.00613",
      "title": "WorldGym full text (arXiv HTML v3): Section 4.1, Table 5, Appendix B.2",
      "type": "paper",
      "date": "2025-09-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://world-model-eval.github.io/abstract.html",
      "title": "WorldGym project page (r = 0.78, 3.3% mean difference)",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2406.09246",
      "title": "OpenVLA (arXiv abs, author list)",
      "type": "paper",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Veo (Robotics)",
    "cat": "own",
    "r": 0.88,
    "n": "8 checkpoints",
    "idx": 12,
    "study": "Veo (Robotics) nominal scenes vs real ALOHA 2 evaluations",
    "statistic": "Pearson = 0.88; MMRV = 0.03 (Figure 4)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.10675",
      "title": "Veo world simulator report full text (arXiv HTML v2): Section 3",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.10675v2/nominal_correlation.png",
      "title": "Veo report Figure 4: nominal real vs predicted success rates",
      "type": "report",
      "date": "2026-01-06",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Runway GWM-Robotics",
    "cat": "own",
    "r": 0.95,
    "n": "8 policies",
    "idx": 16,
    "study": "GWM-Robotics vs RoboArena real-world results",
    "statistic": "Pearson correlation = 0.95 (figure label: Pearson r = 0.952); MMRV = 0.033",
    "sources": [
     {
      "url": "https://runway.com/research/accelerating-robot-policy-evaluation",
      "title": "Accelerating Robot Policy Evaluation with General World Models (Runway Research)",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://d3phaj0sisr2ct.cloudfront.net/research/images/real_vs_gwm1_success_rates-01.png",
      "title": "Runway figure: Real vs. GWM-1 Success Rates",
      "type": "blog",
      "date": "2026-02-27",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Cosmos-Surg-dVRK",
    "cat": "own",
    "r": 0.718,
    "n": "6 checkpoints (human labels)",
    "idx": 18,
    "study": "Cosmos-Surg-dVRK (human labels) vs real dVRK success rates",
    "statistic": "Pooled Pearson r = 0.718 (p < 0.001) across all tasks and training regimes. Per task Pearson / MMRV: Handover 0.468 / 0.217; Throw 0.716 / 0.183; Knot Tie 0.840 / 0.050; Pickup 0.806 / 0.067; average 0.707 / 0.129 (Table 2: Pearson 0.71 +/- 0.17, MMRV 0.13 +/- 0.08). Mean bias error 0.140 (95% CI 0.081-0.199).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2510.16240",
      "title": "Cosmos-Surg-dVRK full text (arXiv HTML v2): Sections 4.1-5.2.1, Tables 1-2",
      "type": "paper",
      "date": "2025-11-03",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "RoboWorld",
    "cat": "own",
    "r": 0.989,
    "n": "8 policies, against RoboArena",
    "idx": 24,
    "study": "RoboWorld vs RoboArena leaderboard",
    "statistic": "Pearson r = 0.989; Spearman rho = 0.970",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.01060",
      "title": "RoboWorld full text (arXiv HTML v4): Sections 1, 5.3",
      "type": "paper",
      "date": "2026-07-15",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2506.18123",
      "title": "RoboArena (arXiv abs, author list)",
      "type": "paper",
      "date": "2025-06-22",
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "DreamDojo",
    "cat": "own",
    "r": 0.995,
    "n": "6 checkpoints of one policy",
    "idx": 27,
    "study": "DreamDojo vs real-robot fruit-packing success",
    "statistic": "Pearson r = 0.995; MMRV = 0.003",
    "sources": [
     {
      "url": "https://arxiv.org/html/2602.06949",
      "title": "DreamDojo full text (arXiv HTML v1), Sec. 4.7 'Downstream Applications' and Limitations",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2602.06949v1/correlation_compressed.svg",
      "title": "DreamDojo Fig. 5(a): real vs DreamDojo success rates (points A-F)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2602.06949",
      "title": "DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-02-06",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "PlayWorld",
    "cat": "own",
    "r": 0.8766,
    "n": "18 policies",
    "idx": 28,
    "study": "PlayWorld vs real-robot success (18 policies)",
    "statistic": "PlayWorld: Pearson r = 0.8766, RMSE = 0.171. Human-demo-trained model: r = 0.6636, RMSE = 0.275. Human-play-trained model: r = 0.6619, RMSE = 0.297 (Fig. 7).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.09030",
      "title": "PlayWorld full text (arXiv HTML v3), Sec. 4.4 and Fig. 7",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v3/images/experiment/correlation_new.png",
      "title": "PlayWorld Fig. 7: policy evaluation success-rate correlation (RMSE and r per training-data source)",
      "type": "paper",
      "date": "2026-04-06",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.09030v1",
      "title": "PlayWorld arXiv HTML v1 (checked that Pearson 0.8766 and 18 policies already appear)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2603.09030",
      "title": "PlayWorld: Learning Robot World Models from Autonomous Play (arXiv abstract; v1 2026-03-09, v3 2026-04-06)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "PersistWorld",
    "cat": "own",
    "r": 0.822,
    "n": "3 policies, 3 tasks",
    "idx": 29,
    "study": "PersistWorld vs real-robot task progress",
    "statistic": "Pearson r = 0.822 (p = 0.007); MMRV = 0.006",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v1",
      "title": "PersistWorld arXiv HTML v1, Appendix 0.C (same Fig. 8 file as v2)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2603.25685",
      "title": "Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning (arXiv abstract; v1 2026-03-26, v2 2026-09-04; ECCV 2026)",
      "type": "paper",
      "date": "2026-03-26",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "WEAVER (fine-tuned)",
    "cat": "own",
    "r": 0.863,
    "n": "2 policies, 5 tasks",
    "idx": 31,
    "study": "WEAVER vs real-robot success (five tasks)",
    "statistic": "WEAVER-FT (Table 8): Pearson = 0.863, Spearman = 0.870, MMRV = 0.035, RMSE = 0.188; abstract and Fig. 6 report rho = 0.870. Pretrained WEAVER: Pearson = 0.563, Spearman = 0.594, MMRV = 0.155, RMSE = 0.359.",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v2/policy_eval.png",
      "title": "WEAVER Fig. 6: policy evaluation scatter plots (rho and MMRV per world model)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2606.13672",
      "title": "WEAVER, Better, Faster, Longer: An Effective World Model for Robotic Manipulation (arXiv abstract; v1 2026-06-11, v2 2026-06-16)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v1",
      "title": "WEAVER arXiv HTML v1 (checked that rho=0.870 and Table 8 values already appear)",
      "type": "paper",
      "date": "2026-06-11",
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "DexTouch-WM (WM-Mix)",
    "cat": "own",
    "r": 0.844,
    "range": [
     0.577,
     0.972
    ],
    "n": "3 policies per task, 4 tasks (mean)",
    "idx": 34,
    "study": "DexTouch-WM vs real dexterous-robot scores",
    "statistic": "Per task (Place Shoes / Place Phone / Stack Bowls / Stand Bottle): WM-Robot Pearson 0.400 / 0.945 / 0.906 / 0.334, MMRV 0.033 / 0 / 0 / 0.067 (mean Pearson 0.646, mean MMRV 0.025); WM-Mix Pearson 0.972 / 0.939 / 0.889 / 0.577, MMRV 0 / 0 / 0 / 0.067 (mean Pearson 0.844, mean MMRV 0.017).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2609.20649",
      "title": "DexTouch-WM full text (arXiv HTML v2), Sec. IV-D, Tables II-III",
      "type": "paper",
      "date": "2026-09-18",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2609.20649",
      "title": "DexTouch-WM: Learning Action-Conditioned Tactile World Models from Human Touch for Dexterous Robot Manipulation (arXiv abstract; v1 2026-09-17, v2 2026-09-18)",
      "type": "paper",
      "date": "2026-09-17",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Interactive World Simulator",
    "cat": "own",
    "range": [
     0.8455,
     0.9908
    ],
    "n": "4 tasks, no overall value",
    "idx": 35,
    "study": "Interactive World Simulator vs real-robot task scores",
    "statistic": "r = 0.8553 (T Pushing), 0.8455 (Rope Routing), 0.8869 (Mug Grasping), 0.9908 (Pile Sweeping) (Fig. 7; the paper labels the coefficient r without naming it)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.08546",
      "title": "Interactive World Simulator full text (arXiv HTML v1), Sec. IV-A, IV-C, IV-D",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.08546v1/correlation_v3.png",
      "title": "Interactive World Simulator Fig. 7: world-simulator vs real task scores per task (r values)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://roboticsproceedings.org/rss22/p018.pdf",
      "title": "RSS 2026 paper PDF p018 (Fig. 7 r values)",
      "type": "paper",
      "date": "2026-07",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2603.08546",
      "title": "Interactive World Simulator for Robot Policy Training and Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-03-09",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "dWorldEval",
    "cat": "own",
    "r": 0.918,
    "n": "3 policies, 5 tasks",
    "idx": 38,
    "study": "dWorldEval vs real AgileX robot success",
    "statistic": "r = 0.918, MMRV = 0.02; no-memory ablation r = 0.829, MMRV = 0.033 (Fig. 7d)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.22152",
      "title": "dWorldEval full text (arXiv HTML v1), Sec. 4.1, 4.3, Fig. 7, Appendix",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2604.22152v1/corr.png",
      "title": "dWorldEval Fig. 7: real vs generated success rates, panels (a)-(d)",
      "type": "paper",
      "date": "2026-04-24",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "OSCAR (skeleton)",
    "cat": "own",
    "r": 0.852,
    "n": "7 policies, against RoboArena",
    "idx": 48,
    "study": "OSCAR vs RoboArena real-world success",
    "statistic": "Skeleton: MMRV 0.571 (scale 0-6), Spearman rho +0.750, Pearson r +0.852, SISR_delta 1.73 pp. Latent action: MMRV 1.429, rho +0.643, r +0.867, SISR_delta 1.98 pp. Mesh: MMRV 0.714, rho +0.679, r +0.781, SISR_delta 3.04 pp (Table 4).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.04463",
      "title": "OSCAR full text (arXiv HTML v2), Sec. 5.4 Table 4, Appendix A.10",
      "type": "paper",
      "date": "2026-06-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2606.04463",
      "title": "OSCAR: Omni-Embodiment Action-Conditioned World Model for Robotics (arXiv abstract; v1 2026-06-03, v2 2026-06-04)",
      "type": "paper",
      "date": "2026-06-03",
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "NVIDIA Cosmos-based model",
    "cat": "own",
    "r": 0.687,
    "n": "3 policies, 4 tasks",
    "idx": 49,
    "study": "Cosmos-based world model vs real Bridge-setup success (NVIDIA)",
    "statistic": "Pearson = 0.687; MMRV = 0.171 (Fig. 6)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520v3/real_world_video_model_correlation_plot_icra.svg",
      "title": "Tseng et al. Fig. 6: policy evaluation on the Bridge setup (Cosmos and IRASim, MMRV and Pearson)",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models (arXiv abstract; v1 2025-11-14, v3 2025-12-04)",
      "type": "paper",
      "date": "2025-11-14",
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Ctrl-World",
    "cat": "other",
    "by": "PersistWorld paper",
    "r": 0.796,
    "n": "3 policies, 3 tasks",
    "idx": 30,
    "study": "Ctrl-World vs real-robot task progress (measured in the PersistWorld paper)",
    "statistic": "Pearson r = 0.796 (p = 0.010); MMRV = 0.053",
    "sources": [
     {
      "url": "https://arxiv.org/html/2603.25685",
      "title": "PersistWorld full text (arXiv HTML v2), 'Policy Evaluation' paragraph and Appendix 0.C",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2603.25685v2/sim_real_comparison_reduced_full_axes_mmrv_fixed.svg",
      "title": "PersistWorld Fig. 8: WM-to-real task progression, Ours vs Baseline (Ctrl-World)",
      "type": "paper",
      "date": "2026-09-04",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "IRASim",
    "cat": "other",
    "by": "NVIDIA paper",
    "r": 0.613,
    "n": "3 policies, 4 tasks",
    "idx": 50,
    "study": "IRASim vs real Bridge-setup success (measured by NVIDIA)",
    "statistic": "Pearson = 0.613; MMRV = 0.611 (Fig. 6)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2511.11520",
      "title": "Scalable Policy Evaluation with Video World Models full text (arXiv HTML v3), Table I, Sec. IV-C, IV-D",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/html/2511.11520v3/real_world_video_model_correlation_plot_icra.svg",
      "title": "Tseng et al. Fig. 6: policy evaluation on the Bridge setup (Cosmos and IRASim, MMRV and Pearson)",
      "type": "paper",
      "date": "2025-12-04",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2406.14540",
      "title": "IRASim: A Fine-Grained World Model for Robot Manipulation (arXiv abstract; v1 2024-06-20, v2 2025-07-29)",
      "type": "paper",
      "date": "2025-07-29",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Ctrl-World",
    "cat": "other",
    "by": "WEAVER paper",
    "r": 0.552,
    "n": "2 policies, 5 tasks",
    "idx": 32,
    "study": "Ctrl-World vs real-robot success (measured in the WEAVER paper)",
    "statistic": "Pearson = 0.552; Spearman = 0.523; MMRV = 0.215; RMSE = 0.410 (Table 8; Fig. 6 shows rho = 0.523, MMRV = 0.215)",
    "sources": [
     {
      "url": "https://arxiv.org/html/2606.13672",
      "title": "WEAVER full text (arXiv HTML v2), Sec. 3.4, 4, 5.2.1, Appendix A4.1 Table 8",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2606.13672v2/policy_eval.png",
      "title": "WEAVER Fig. 6: policy evaluation scatter plots (rho and MMRV per world model)",
      "type": "paper",
      "date": "2026-06-16",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation (arXiv abstract; author list)",
      "type": "paper",
      "date": "2025-10-11",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "wm": "Ctrl-World",
    "cat": "other",
    "by": "PolaRiS paper, which shares an author with Ctrl-World",
    "r": 0.53,
    "n": "4 policies",
    "idx": 9,
    "study": "Ctrl-World as a baseline evaluator in the PolaRiS study",
    "statistic": "Pearson r = 0.53; MMRV = 0.22 (PolaRiS Figure 7). Same figure, other evaluators: PolaRiS r = 0.90, MMRV = 0.03; LIBERO-90 fine-tuned checkpoints at 1k / 10k / 50k steps r = 0.66 / 0.70 / 0.66, MMRV = 0.19 / 0.04 / 0.15; action MSE r = -0.55 (train) and -0.53 (validation), MMRV = 0.40 for both.",
    "sources": [
     {
      "url": "https://arxiv.org/html/2512.16881",
      "title": "PolaRiS full text (arXiv HTML v2): Sections 5.1-5.2",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2512.16881v2/figures/main_barplots.png",
      "title": "PolaRiS Figure 7: Pearson r and MMRV by evaluation method",
      "type": "paper",
      "date": "2025-12-30",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2510.10125",
      "title": "Ctrl-World (arXiv abs, author list)",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   }
  ],
  "no_coefficient": [
   {
    "name": "Ctrl-World vs real-robot rollouts on the authors' DROID setup",
    "statistic": "No correlation coefficient or MMRV reported. Linear fits of world-model rate on real rate across policy-task pairs (Figure 7): instruction following y = 0.87x - 0.04; success rate y = 0.81x - 0.11.",
    "sources": [
     {
      "url": "https://arxiv.org/pdf/2510.10125v3",
      "title": "Ctrl-World PDF v3: Section 5.3, Figure 7",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2510.10125",
      "title": "Ctrl-World full text (arXiv HTML v3): Appendix B Table 3",
      "type": "paper",
      "date": "2026-03-01",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://github.com/Robert-gyj/Ctrl-World",
      "title": "Ctrl-World repository readme (20 runs per task category)",
      "type": "repo",
      "date": "2025-10-09",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "name": "1X World Model outcome-prediction alignment",
    "statistic": "Alignment 63.06% when trained on about 216M Shelf video tokens; 71.17% with an added about 1.46B Arcade video tokens",
    "sources": [
     {
      "url": "https://www.1x.tech/1x-world-model.pdf",
      "title": "1X World Model: Evaluating Bits, not Atoms, Sections 5.1-5.3",
      "type": "report",
      "date": "2025-06-16",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "name": "GigaWorld-1 closed-loop success-rate alignment with real robots",
    "statistic": "No correlation statistic. Fitted line of generated vs real success rate: y = 1.134x - 0.091 (Fig. 16). Generated minus real success per subtask ranges from -0.13 to +0.07 across 12 subtasks (Fig. 17).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2607.02642",
      "title": "GigaWorld-1 full text (arXiv HTML v1), Sec. 4 (WMBench), 5.1, 6.5",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/images/multi_task_success_rate_fit.png",
      "title": "GigaWorld-1 Fig. 16: real vs generated success rate with fit lines",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2607.02642v1/success_rate_difference_bar.svg",
      "title": "GigaWorld-1 Fig. 17: Gen - Real success-rate difference per subtask and model",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/abs/2607.02642",
      "title": "GigaWorld-1: A Roadmap to Build World Models for Robot Policy Evaluation (arXiv abstract, v1)",
      "type": "paper",
      "date": "2026-07-02",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "name": "EnerVerse-AC vs real-robot success (four tasks, three training steps)",
    "statistic": "No statistic; bar charts show the same task ranking and the same training-step trend. Real vs EVAC per task: Take a Bottle 28% vs 25%; Take a Toast 100% vs 90%; Take a Bacon 85% vs 88%; Take a Leaf 55% vs 50%. Per training step (Take a Bottle): 4K 40% vs 40%; 8K 61% vs 63%; 13K 79% vs 76% (Fig. 7).",
    "sources": [
     {
      "url": "https://arxiv.org/html/2505.09723",
      "title": "EnerVerse-AC full text (arXiv HTML v1), Sec. 3 'Evaluator for Policy Model', Sec. 4.3, Appendix A.4.1",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     },
     {
      "url": "https://arxiv.org/html/2505.09723v1/Fig5_Comp_Real_CAE.svg",
      "title": "EnerVerse-AC Fig. 7: success rate per task and per learning step, real robot vs EVAC",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://arxiv.org/abs/2505.09723",
      "title": "EnerVerse-AC: Envisioning Embodied Environments with Action Condition (arXiv abstract, v1)",
      "type": "paper",
      "date": "2025-05-14",
      "accessed": "2026-10-10"
     }
    ],
    "level": "verified"
   },
   {
    "name": "RoboWM-Bench real-to-sim outcome consistency",
    "statistic": "Success consistency 10/10 and failure consistency 10/10 for every task (Pick Object, Pull Object, Push Object, Put on Plate, Discard Trash, Close Drawer, Put in Drawer): 140 of 140 outcomes matched",
    "sources": [
     {
      "url": "https://arxiv.org/html/2604.19092v2",
      "title": "RoboWM-Bench full text (arXiv HTML v2)",
      "type": "paper",
      "date": "2026-05-14",
      "accessed": "2026-10-11"
     },
     {
      "url": "https://robowm-bench.github.io/RoboWM-Bench/",
      "title": "RoboWM-Bench project page",
      "type": "project-page",
      "date": null,
      "accessed": "2026-10-11"
     }
    ],
    "level": "verified"
   }
  ]
 }
}