[
  {
    "id": "world-models-2018",
    "name": "World Models",
    "org": "unknown",
    "route": "latent",
    "one_liner": "A latent environment model used to train a controller inside imagined rollouts.",
    "input": "unknown",
    "output": "Predicted latent-state rollouts used by a controller.",
    "action_conditioned": "unknown",
    "public": "unknown",
    "open_source": "unknown",
    "limit": "unknown",
    "confused_with": [],
    "sources": [
      "https://worldmodels.github.io/"
    ],
    "updated": "2026-07-09"
  },
  {
    "id": "sora-world-simulators",
    "name": "Sora",
    "org": "OpenAI",
    "route": "pixel",
    "one_liner": "A video-generation model described in an official technical note about video models as world simulators.",
    "input": "unknown",
    "output": "Minute-scale generated video.",
    "action_conditioned": "unknown",
    "public": "unknown",
    "open_source": "unknown",
    "limit": "unknown",
    "confused_with": [
      "interactive world model"
    ],
    "sources": [
      "https://openai.com/index/video-generation-models-as-world-simulators/"
    ],
    "updated": "2026-07-09"
  },
  {
    "id": "copilot4d",
    "name": "Copilot4D",
    "org": "Waabi",
    "route": "latent",
    "one_liner": "Waabi's research page describes a driving world model that tokenizes lidar point clouds and predicts future observations with discrete diffusion.",
    "input": "Lidar point clouds and a candidate future action.",
    "output": "Predicted future lidar point clouds.",
    "action_conditioned": true,
    "public": "unknown",
    "open_source": "unknown",
    "limit": "unknown",
    "confused_with": [
      "camera-video generator",
      "driving policy"
    ],
    "sources": [
      "https://waabi.ai/research/copilot-4d"
    ],
    "updated": "2026-07-27"
  },
  {
    "id": "genie-3",
    "name": "Genie 3",
    "org": "Google DeepMind",
    "route": "interactive",
    "one_liner": "Google DeepMind describes a world model that generates dynamic environments from text and supports real-time navigation.",
    "input": "Text prompts and user navigation actions.",
    "output": "Dynamic visual environments that update during navigation.",
    "action_conditioned": true,
    "public": "limited_access",
    "open_source": "unknown",
    "limit": "unknown",
    "confused_with": [
      "passive video generator"
    ],
    "sources": [
      "https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/"
    ],
    "updated": "2026-07-09"
  },
  {
    "id": "vjepa-2-ac",
    "name": "V-JEPA 2-AC",
    "org": "Meta AI",
    "route": "latent",
    "one_liner": "An action-conditioned latent predictor post-trained from a frozen V-JEPA 2 encoder for robot planning.",
    "input": "Robot video, end-effector state, actions, and an image goal.",
    "output": "Predicted future representations used for model-predictive control.",
    "action_conditioned": true,
    "public": "open_weights",
    "open_source": "unknown",
    "limit": "The action-conditioned controller is separately post-trained, and the retained physical-robot results are developer-run.",
    "confused_with": [
      "V-JEPA 2 base video-representation model",
      "pixel generator"
    ],
    "sources": [
      "https://ai.meta.com/blog/v-jepa-2-world-model-benchmarks/",
      "https://arxiv.org/abs/2506.09985v1"
    ],
    "updated": "2026-07-12"
  },
  {
    "id": "marble",
    "name": "Marble",
    "org": "World Labs",
    "route": "interactive",
    "one_liner": "World Labs describes a multimodal system for creating, editing, expanding, combining, and exporting 3D worlds.",
    "input": "Text, images, video, or coarse 3D layouts.",
    "output": "Editable 3D worlds exportable as Gaussian splats, meshes, or videos.",
    "action_conditioned": "unknown",
    "public": "public",
    "open_source": "unknown",
    "limit": "The official product post describes its collider mesh as low-fidelity and intended for coarse physics.",
    "confused_with": [
      "metric 3D reconstruction",
      "physics simulator"
    ],
    "sources": [
      "https://www.worldlabs.ai/blog/marble-world-model"
    ],
    "updated": "2026-07-28"
  },
  {
    "id": "gwm-1",
    "name": "GWM-1",
    "org": "Runway",
    "route": "interactive",
    "one_liner": "Runway describes an autoregressive real-time model family with action control across worlds, avatars, and robotics variants.",
    "input": "Camera pose, robot commands, or audio, depending on the separately post-trained variant.",
    "output": "Frame-by-frame generated visual rollouts.",
    "action_conditioned": true,
    "public": "application_required",
    "open_source": "unknown",
    "limit": "GWM Worlds, GWM Avatars, and GWM Robotics are three separately post-trained variants rather than one unified model.",
    "confused_with": [
      "one unified general model"
    ],
    "sources": [
      "https://runway.com/research/introducing-runway-gwm-1"
    ],
    "updated": "2026-07-27"
  },
  {
    "id": "waymo-world-model",
    "name": "Waymo World Model",
    "org": "Waymo",
    "route": "interactive",
    "one_liner": "Waymo describes a Genie 3-derived driving simulator post-trained to generate camera and lidar output under multiple controls.",
    "input": "Dashcam video, driving actions, scene layout, and language controls.",
    "output": "Camera and lidar simulation rollouts.",
    "action_conditioned": true,
    "public": "demo_only",
    "open_source": "unknown",
    "limit": "unknown",
    "confused_with": [
      "validated safety model"
    ],
    "sources": [
      "https://waymo.com/blog/2026/02/the-waymo-world-model-a-new-frontier-for-autonomous-driving-simulation/"
    ],
    "updated": "2026-07-27"
  },
  {
    "id": "cosmos-3",
    "name": "Cosmos 3",
    "org": "NVIDIA",
    "route": "physics",
    "one_liner": "NVIDIA describes an omnimodal physical-AI model combining vision reasoning, world generation, and action prediction.",
    "input": "Language, video, action, and audio signals.",
    "output": "Vision-reasoning results, generated world rollouts, and action predictions.",
    "action_conditioned": "unknown",
    "public": "open_weights",
    "open_source": true,
    "limit": "unknown",
    "confused_with": [
      "fully open-source training stack"
    ],
    "sources": [
      "https://investor.nvidia.com/news/press-release-details/2026/NVIDIA-Launches-Cosmos-3-the-Open-Frontier-Foundation-Model-for-Physical-AI/default.aspx",
      "https://arxiv.org/abs/2606.02800v4"
    ],
    "updated": "2026-08-06"
  },
  {
    "id": "gaia-4",
    "name": "GAIA-4",
    "org": "Wayve",
    "route": "interactive",
    "one_liner": "Wayve describes a driving world model used in closed-loop simulation with its AI Driver in the loop.",
    "input": "Recorded sensor data and the AI Driver's actions.",
    "output": "Generated camera and radar inputs for the driving policy.",
    "action_conditioned": true,
    "public": "demo_only",
    "open_source": "unknown",
    "limit": "In world-on-rails mode, other road users remain on logged trajectories instead of reacting to the ego vehicle.",
    "confused_with": [
      "road-test replacement",
      "fully reactive traffic simulation in world-on-rails mode"
    ],
    "sources": [
      "https://wayve.ai/thinking/gaia-4/"
    ],
    "updated": "2026-08-05"
  }
]
