{ "video_generation": [ { "title": "PhyWorld: Physics-Faithful World Model for Video Generation", "authors": [ "Pu Zhao", "Juyi Lin", "Timothy Rupprecht", "Arash Akbari", "Chence Yang" ], "summary": "World simulators can provide safe and scalable environments for training Physical AI systems before real-world deployment. Large video generation models are emerging as a promising basis for such simulators because they can generate diverse and realistic visual futures. However, using them as world simulators requires physically faithful video continuations, namely, generated videos that preserve the physical state implied by the conditioning input, and evolve in ways consistent with basic physi...", "published": "2026-05-19", "arxiv_id": "2605.19242", "link": "https://arxiv.org/abs/2605.19242", "categories": [ "cs.CV", "cs.AI", "cs.ET", "cs.LG", "cs.MM" ] }, { "title": "ACWM-Phys: Investigating Generalized Physical Interaction in Action-Conditioned Video World Models", "authors": [ "Haotian Xue", "Yipu Chen", "Liqian Ma", "Zelin Zhao", "Lama Moukheiber" ], "summary": "Action-conditioned world models (ACWMs) have shown strong promise for video prediction and decision-making. However, existing benchmarks are largely restricted to egocentric navigation or narrow, task-specific robotics datasets, offering only limited coverage of the rich physical interactions required for generalized world understanding. We introduce ACWM-Phys, a new benchmark for evaluating action-conditioned prediction under diverse physical dynamics in a clean, controllable simulation environ...", "published": "2026-05-09", "arxiv_id": "2605.08567", "link": "https://arxiv.org/abs/2605.08567", "categories": [ "cs.CV" ] }, { "title": "Video Generation Models as World Models: Efficient Paradigms, Architectures and Algorithms", "authors": [ "Muyang He", "Hanzhong Guo", "Junxiong Lin", "Yizhou Yu" ], "summary": "The rapid evolution of video generation has enabled models to simulate complex physical dynamics and long-horizon causalities, positioning them as potential world simulators. However, a critical gap still remains between the theoretical capacity for world simulation and the heavy computational costs of spatiotemporal modeling. To address this, we comprehensively and systematically review video generation frameworks and techniques that consider efficiency as a crucial requirement for practical wo...", "published": "2026-03-30", "arxiv_id": "2603.28489", "link": "https://arxiv.org/abs/2603.28489", "categories": [ "eess.IV", "cs.CV" ] }, { "title": "Stereo World Model: Camera-Guided Stereo Video Generation", "authors": [ "Yang-Tian Sun", "Zehuan Huang", "Yifan Niu", "Lin Ma", "Yan-Pei Cao" ], "summary": "We present StereoWorld, a camera-conditioned stereo world model that jointly learns appearance and binocular geometry for end-to-end stereo video generation.Unlike monocular RGB or RGBD approaches, StereoWorld operates exclusively within the RGB modality, while simultaneously grounding geometry directly from disparity. To efficiently achieve consistent stereo generation, our approach introduces two key designs: (1) a unified camera-frame RoPE that augments latent tokens with camera-aware rotary ...", "published": "2026-03-18", "arxiv_id": "2603.17375", "link": "https://arxiv.org/abs/2603.17375", "categories": [ "cs.CV" ] }, { "title": "SAW: Toward a Surgical Action World Model via Controllable and Scalable Video Generation", "authors": [ "Sampath Rapuri", "Lalithkumar Seenivasan", "Dominik Schneider", "Roger Soberanis-Mukul", "Yufan He" ], "summary": "A surgical world model capable of generating realistic surgical action videos with precise control over tool-tissue interactions can address fundamental challenges in surgical AI and simulation -- from data scarcity and rare event synthesis to bridging the sim-to-real gap for surgical automation. However, current video generation methods, the very core of such surgical world models, require expensive annotations or complex structured intermediates as conditioning signals at inference, limiting t...", "published": "2026-03-13", "arxiv_id": "2603.13024", "link": "https://arxiv.org/abs/2603.13024", "categories": [ "cs.CV", "cs.AI", "cs.LG", "eess.IV" ] }, { "title": "LiveWorld: Simulating Out-of-Sight Dynamics in Generative Video World Models", "authors": [ "Zicheng Duan", "Jiatong Xia", "Zeyu Zhang", "Wenbo Zhang", "Gengze Zhou" ], "summary": "Recent generative video world models aim to simulate visual environment evolution, allowing an observer to interactively explore the scene via camera control. However, they implicitly assume that the world only evolves within the observer's field of view. Once an object leaves the observer's view, its state is \"frozen\" in memory, and revisiting the same region later often fails to reflect events that should have occurred in the meantime. In this work, we identify and formalize this overlooked li...", "published": "2026-03-07", "arxiv_id": "2603.07145", "link": "https://arxiv.org/abs/2603.07145", "categories": [ "cs.CV" ] }, { "title": "ShareVerse: Multi-Agent Consistent Video Generation for Shared World Modeling", "authors": [ "Jiayi Zhu", "Jianing Zhang", "Yiying Yang", "Wei Cheng", "Xiaoyun Yuan" ], "summary": "This paper presents ShareVerse, a video generation framework enabling multi-agent shared world modeling, addressing the gap in existing works that lack support for unified shared world construction with multi-agent interaction. ShareVerse leverages the generation capability of large video models and integrates three key innovations: 1) A dataset for large-scale multi-agent interactive world modeling is built on the CARLA simulation platform, featuring diverse scenes, weather conditions, and inte...", "published": "2026-03-03", "arxiv_id": "2603.02697", "link": "https://arxiv.org/abs/2603.02697", "categories": [ "cs.CV", "cs.AI" ] }, { "title": "DreamWorld: Unified World Modeling in Video Generation", "authors": [ "Boming Tan", "Xiangdong Zhang", "Ning Liao", "Yuqing Zhang", "Shaofeng Zhang" ], "summary": "Despite impressive progress in video generation, existing models remain limited to surface-level plausibility, lacking a coherent and unified understanding of the world. Prior approaches typically incorporate only a single form of world-related knowledge or rely on rigid alignment strategies to introduce additional knowledge. However, aligning the single world knowledge is insufficient to constitute a world model that requires jointly modeling multiple heterogeneous dimensions (e.g., physical co...", "published": "2026-02-28", "arxiv_id": "2603.00466", "link": "https://arxiv.org/abs/2603.00466", "categories": [ "cs.CV" ] } ], "autonomous_driving": [ { "title": "HEAT: Heterogeneous End-to-End Autonomous Driving via Trajectory-Guided World Models", "authors": [ "Hoonhee Cho", "Giwon Lee", "Jae-Young Kang", "Hyemin Yang", "Heejun Park" ], "summary": "End-to-end autonomous driving has emerged as a compelling alternative to traditional modular pipelines by directly mapping raw sensor data to driving actions. While recent approaches achieve strong performance on single-domain datasets, their performance degrades significantly when trained jointly across multiple heterogeneous domains. In practice, however, autonomous systems must operate across diverse environments with heterogeneous distributions, including different cities, sensor configurati...", "published": "2026-05-19", "arxiv_id": "2605.19631", "link": "https://arxiv.org/abs/2605.19631", "categories": [ "cs.RO", "cs.CV" ] }, { "title": "Xiaomi EV World Model: A Joint World Model Integrating Reconstruction and Generation for Autonomous Driving", "authors": [ "Lijun Zhou", "Hongcheng Luo", "Zhenxin Zhu", "Cheng Chi", "Mingfei Tu" ], "summary": "This report presents a unified technical system addressing the two core capabilities of world models for autonomous driving: world representation and world generation. For world representation, we propose WorldRec, a feed-forward reconstruction architecture driven by sparse scene queries. WorldRec initializes structured queries in 3D space, leveraging them to aggregate cross-view, cross-temporal features, thereby naturally enforcing spatial consistency across frames and yielding compact yet high...", "published": "2026-05-18", "arxiv_id": "2605.18137", "link": "https://arxiv.org/abs/2605.18137", "categories": [ "cs.CV" ] }, { "title": "DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving", "authors": [ "Lingjun Zhang", "Changjie Wu", "Linzhe Shi", "Jiangyang Li", "Jiaxin Liu" ], "summary": "End-to-end autonomous driving systems are increasingly integrating Vision-Language Model (VLM) architectures, incorporating text reasoning or visual reasoning to enhance the robustness and accuracy of driving decisions. However, the reasoning mechanisms employed in most methods are direct adaptations from general domains, lacking in-depth exploration tailored to autonomous driving scenarios, particularly within visual reasoning modules. In this paper, we propose a driving world model that perfor...", "published": "2026-05-11", "arxiv_id": "2605.10564", "link": "https://arxiv.org/abs/2605.10564", "categories": [ "cs.CV", "cs.RO" ] }, { "title": "CoWorld-VLA: Thinking in a Multi-Expert World Model for Autonomous Driving", "authors": [ "Minqing Huang", "Yujiao Xiang", "Zihan Liang", "Jiajie Huang", "Jingqi Wang" ], "summary": "Vision-Language-Action (VLA) models have emerged as a promising paradigm for end-to-end autonomous driving. However, existing reasoning mechanisms still struggle to provide planning-oriented intermediate representations: textual Chain-of-Thought (CoT) fails to preserve continuous spatiotemporal structure, while latent world reasoning remains difficult to use as a direct condition for action generation. In this paper, we propose CoWorld-VLA, a multi-expert world reasoning framework for autonomous...", "published": "2026-05-11", "arxiv_id": "2605.10426", "link": "https://arxiv.org/abs/2605.10426", "categories": [ "cs.CV", "cs.AI" ] }, { "title": "DriveFuture: Future-Aware Latent World Models for Autonomous Driving", "authors": [ "Yufeng Hong", "Xiaotian Zhou", "Yingyan Li", "Xiangpo Zhou", "Lin Liu" ], "summary": "Existing latent world models for autonomous driving have opened a promising path toward future-aware driving intelligence. However, they typically treat future latent states as prediction targets or auxiliary signals, rather than directly conditioning trajectory planning. This can entangle current and future features in latent space. In this work, we propose DriveFuture, a future-aware latent world modeling framework for autonomous driving that explicitly learns planning-oriented foresight by co...", "published": "2026-05-10", "arxiv_id": "2605.09701", "link": "https://arxiv.org/abs/2605.09701", "categories": [ "cs.CV" ] }, { "title": "Learning Vision-Language-Action World Models for Autonomous Driving", "authors": [ "Guoqing Wang", "Pin Tang", "Xiangxuan Ren", "Guodongfang Zhao", "Bailan Feng" ], "summary": "Vision-Language-Action (VLA) models have recently achieved notable progress in end-to-end autonomous driving by integrating perception, reasoning, and control within a unified multimodal framework. However, they often lack explicit modeling of temporal dynamics and global world consistency, which limits their foresight and safety. In contrast, world models can simulate plausible future scenes but generally struggle to reason about or evaluate the imagined future they generate. In this work, we p...", "published": "2026-04-10", "arxiv_id": "2604.09059", "link": "https://arxiv.org/abs/2604.09059", "categories": [ "cs.CV", "cs.AI" ] }, { "title": "ExploreVLA: Dense World Modeling and Exploration for End-to-End Autonomous Driving", "authors": [ "Zihao Sheng", "Xin Ye", "Jingru Luo", "Sikai Chen", "Liu Ren" ], "summary": "End-to-end autonomous driving models based on Vision-Language-Action (VLA) architectures have shown promising results by learning driving policies through behavior cloning on expert demonstrations. However, imitation learning inherently limits the model to replicating observed behaviors without exploring diverse driving strategies, leaving it brittle in novel or out-of-distribution scenarios. Reinforcement learning (RL) offers a natural remedy by enabling policy exploration beyond the expert dis...", "published": "2026-04-03", "arxiv_id": "2604.02714", "link": "https://arxiv.org/abs/2604.02714", "categories": [ "cs.CV" ] }, { "title": "DLWM: Dual Latent World Models enable Holistic Gaussian-centric Pre-training in Autonomous Driving", "authors": [ "Yiyao Zhu", "Ying Xue", "Haiming Zhang", "Guangfeng Jiang", "Wending Zhou" ], "summary": "Vision-based autonomous driving has gained much attention due to its low costs and excellent performance. Compared with dense BEV (Bird's Eye View) or sparse query models, Gaussian-centric method is a comprehensive yet sparse representation by describing scene with 3D semantic Gaussians. In this paper, we introduce DLWM, a novel paradigm with Dual Latent World Models specifically designed to enable holistic gaussian-centric pre-training in autonomous driving using two stages. In the first stage,...", "published": "2026-04-01", "arxiv_id": "2604.00969", "link": "https://arxiv.org/abs/2604.00969", "categories": [ "cs.CV" ] } ], "embodied_ai": [ { "title": "World-Ego Modeling for Long-Horizon Evolution in Hybrid Embodied Tasks", "authors": [ "Zuyao Lin", "Jianhui Zhang", "Peidong Jia", "Xiaoguang Zhao", "Shanghang Zhang" ], "summary": "World models are widely explored in embodied intelligence, yet they typically predict distinct evolutions of the world and the ego within a single stream, where the world captures persistent instruction-agnostic scene regularities and the ego captures robot-centric instruction-conditioned dynamics. This world-ego entanglement leads to a degradation in long-horizon embodied scenarios, particularly in hybrid tasks with interleaved navigation and manipulation behaviors. In this paper, we introduce ...", "published": "2026-05-19", "arxiv_id": "2605.19957", "link": "https://arxiv.org/abs/2605.19957", "categories": [ "cs.CV", "cs.AI", "cs.RO" ] }, { "title": "SWEET: Sparse World Modeling with Image Editing for Embodied Task Execution", "authors": [ "Yiren Song", "Yihan Wang", "Xiyao Deng", "Zhuoran Yan", "Mike Zheng Shou" ], "summary": "Visual prediction has emerged as a promising paradigm for embodied control, where future observations are generated and then translated into actions. However, dense video generation is computationally expensive and often unnecessary for many manipulation tasks, whose progress can be summarized by a small number of task-relevant visual states. In this work, we study whether image editing models can serve as sparse visual world models for robot manipulation by predicting task-level future states w...", "published": "2026-05-19", "arxiv_id": "2605.19319", "link": "https://arxiv.org/abs/2605.19319", "categories": [ "cs.CV" ] }, { "title": "Key-Gram: Extensible World Knowledge for Embodied Manipulation", "authors": [ "Jingjing Fan", "Siyuan Li", "Botao Ren", "Zhidong Deng" ], "summary": "Embodied control increasingly requires models to follow compositional language instructions while reasoning over dynamic visual states. However, current vision-language-action policies and world-action models often couple linguistic knowledge with visual computation in a shared backbone or conditioning pathway, leading to modality competition and making knowledge extension dependent on backbone updates. In this paper, we introduce Key-Gram, a conditional-memory framework that separates language-...", "published": "2026-05-18", "arxiv_id": "2605.18556", "link": "https://arxiv.org/abs/2605.18556", "categories": [ "cs.RO", "cs.AI" ] }, { "title": "WorldArena 2.0: Extending Embodied World Model Benchmarking on Modality, Functionality and Platform", "authors": [ "Yu Shang", "Yinzhou Tang", "Yiding Ma", "Zhuohang Li", "Lei Jin" ], "summary": "World models have emerged as a central paradigm for embodied intelligence, enabling agents to predict action-conditioned future and reason about environmental dynamics. However, existing embodied world model benchmarks are still largely confined to vision-only prediction, offline embodied applications, and simulator-based evaluation, making them insufficient for assessing increasingly comprehensive world models. In this work, we introduce WorldArena 2.0, an expanded benchmark that systematically...", "published": "2026-05-18", "arxiv_id": "2605.17912", "link": "https://arxiv.org/abs/2605.17912", "categories": [ "cs.RO", "cs.CV" ] }, { "title": "RoboFlow4D: A Lightweight Flow World Model Toward Real-Time Flow-Guided Robotic Manipulation", "authors": [ "Sixu Lin", "Junliang Chen", "Huaiyuan Xu", "Zhuohao Li", "Guangming Wang" ], "summary": "Planning and acting in 3D environments is a fundamental capability for robotic manipulation in the real world. Although prior work has explored predictive flow planners to guide 3D manipulation, existing approaches often rely on modular pipelines stacking multiple submodels, resulting in high computational overhead and limited real-time performance. To address these challenges, we introduce RoboFlow4D, a lightweight flow world model that unifies perception and planning by estimating temporal mot...", "published": "2026-05-17", "arxiv_id": "2605.17522", "link": "https://arxiv.org/abs/2605.17522", "categories": [ "cs.RO" ] }, { "title": "DeTrack: A Benchmark and Altitude-Aware Dual World Model for Drone-embodied Tracking", "authors": [ "Guyue Hu", "Haoming Liu", "Siyuan Song", "Chenglong Li", "Feng Chen" ], "summary": "Aerial object tracking has broad applications in public safety, emergency rescue, wildlife monitoring, and related fields. However, existing aerial tracking benchmarks are mainly based on passive 2D video sequences captured from fixed camera locations or predefined flight paths, where drones are treated as passive cameras rather than embodied agents that actively perceive, interact, and control their motion in dynamic 3D scenes. In this paper, we define a new drone-embodied tracking task, termed...", "published": "2026-05-17", "arxiv_id": "2605.17451", "link": "https://arxiv.org/abs/2605.17451", "categories": [ "cs.CV" ] }, { "title": "Embodied Multi-Agent Coordination by Aligning World Models Through Dialogue", "authors": [ "Vardhan Dongre", "Dilek Hakkani-Tür" ], "summary": "Effective collaboration between embodied agents requires more than acting in a shared environment; it demands communication grounded in each agent's evolving understanding of the world. When agents can only partially observe their surroundings, coordination without communication is provably hard, but communication can, in principle, bridge this gap by allowing agents to share observations and align their world models. In this work, we examine whether LLM-based embodied agents actually realize th...", "published": "2026-05-13", "arxiv_id": "2605.12920", "link": "https://arxiv.org/abs/2605.12920", "categories": [ "cs.MA", "cs.AI", "cs.CL" ] }, { "title": "OrbiSim: World Models as Differentiable Physics Engines for Embodied Intelligence", "authors": [ "Jiajian Li", "Jingyuan Huang", "Junru Gong", "Qi Wang", "Xiaokang Yang" ], "summary": "We present OrbiSim, a novel robotic simulation paradigm that redefines world models as a fully differentiable physics engine for embodied intelligence. Unlike prior world models that focus on unconstrained imagination in latent or visual domains, OrbiSim establishes a unified, physically-grounded pathway that bridges structured scene assets, neural dynamics, and downstream reinforcement learning. By enabling end-to-end differentiability throughout the entire simulation loop -- spanning from expl...", "published": "2026-05-12", "arxiv_id": "2605.16395", "link": "https://arxiv.org/abs/2605.16395", "categories": [ "cs.RO", "cs.LG" ] } ], "reinforcement_learning": [ { "title": "JEDI: Joint Embedding Diffusion World Model for Online Model-Based Reinforcement Learning", "authors": [ "Jing Yu Lim", "Rushi Shah", "Zarif Ikram", "Samson Yu", "Haozhe Ma" ], "summary": "Diffusion world models have recently become competitive for online model-based reinforcement learning, but current approaches expose a tension: pixel diffusion is effective but computationally expensive while the latest latent diffusion approach improves efficiency yet performs subpar. The latter also relies on separately trained latents rather than the end-to-end world-model objectives that have driven much of modern MBRL progress. In particular, JEPA-style predictive representation learning ha...", "published": "2026-05-13", "arxiv_id": "2605.13013", "link": "https://arxiv.org/abs/2605.13013", "categories": [ "cs.LG" ] }, { "title": "WOMBET: World Model-based Experience Transfer for Robust and Sample-efficient Reinforcement Learning", "authors": [ "Mintae Kim", "Koushil Sreenath" ], "summary": "Reinforcement learning (RL) in robotics is often limited by the cost and risk of data collection, motivating experience transfer from a source task to a target task. Offline-to-online RL leverages prior data but typically assumes a given fixed dataset and does not address how to generate reliable data for transfer. We propose \\textit{World Model-based Experience Transfer} (WOMBET), a framework that jointly generates and utilizes prior data. WOMBET learns a world model in the source task and gene...", "published": "2026-04-10", "arxiv_id": "2604.08958", "link": "https://arxiv.org/abs/2604.08958", "categories": [ "cs.LG", "cs.AI", "cs.RO" ] }, { "title": "Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning", "authors": [ "Jai Bardhan", "Patrik Drozdik", "Josef Sivic", "Vladimir Petrik" ], "summary": "Action-conditioned robot world models generate future video frames of the manipulated scene given a robot action sequence, offering a promising alternative for simulating tasks that are difficult to model with traditional physics engines. However, these models are optimized for short-term prediction and break down when deployed autoregressively: each predicted clip feeds back as context for the next, causing errors to compound and visual quality to rapidly degrade. We address this through the fo...", "published": "2026-03-26", "arxiv_id": "2603.25685", "link": "https://arxiv.org/abs/2603.25685", "categories": [ "cs.RO", "cs.CV" ] }, { "title": "DreamerAD: Efficient Reinforcement Learning via Latent World Model for Autonomous Driving", "authors": [ "Pengxuan Yang", "Yupeng Zheng", "Deheng Qian", "Zebin Xing", "Qichao Zhang" ], "summary": "We introduce DreamerAD, the first latent world model framework that enables efficient reinforcement learning for autonomous driving by compressing diffusion sampling from 100 steps to 1 - achieving 80x speedup while maintaining visual interpretability. Training RL policies on real-world driving data incurs prohibitive costs and safety risks. While existing pixel-level diffusion world models enable safe imagination-based training, they suffer from multi-step diffusion inference latency (2s/frame)...", "published": "2026-03-25", "arxiv_id": "2603.24587", "link": "https://arxiv.org/abs/2603.24587", "categories": [ "cs.LG", "cs.RO" ] }, { "title": "Model Predictive Control with Differentiable World Models for Offline Reinforcement Learning", "authors": [ "Rohan Deb", "Stephen J. Wright", "Arindam Banerjee" ], "summary": "Offline Reinforcement Learning (RL) aims to learn optimal policies from fixed offline datasets, without further interactions with the environment. Such methods train an offline policy (or value function), and apply it at inference time without further refinement. We introduce an inference time adaptation framework inspired by model predictive control (MPC) that utilizes a pretrained policy along with a learned world model of state transitions and rewards. While existing world model and diffusion...", "published": "2026-03-23", "arxiv_id": "2603.22430", "link": "https://arxiv.org/abs/2603.22430", "categories": [ "cs.LG" ] }, { "title": "Towards Practical World Model-based Reinforcement Learning for Vision-Language-Action Models", "authors": [ "Zhilong Zhang", "Haoxiang Ren", "Yihao Sun", "Yifei Sheng", "Haonan Wang" ], "summary": "Vision-Language-Action (VLA) models show strong generalization for robotic control, but finetuning them with reinforcement learning (RL) is constrained by the high cost and safety risks of real-world interaction. Training VLA models in interactive world models avoids these issues but introduces several challenges, including pixel-level world modeling, multi-view consistency, and compounding errors under sparse rewards. Building on recent advances across large multimodal models and model-based RL...", "published": "2026-03-21", "arxiv_id": "2603.20607", "link": "https://arxiv.org/abs/2603.20607", "categories": [ "cs.RO", "cs.LG" ] }, { "title": "AcceRL: A Distributed Asynchronous Reinforcement Learning and World Model Framework for Vision-Language-Action Models", "authors": [ "Chengxuan Lu", "Shukuan Wang", "Yanjie Li", "Wei Liu", "Shiji Jin" ], "summary": "Reinforcement learning (RL) for large-scale Vision-Language-Action (VLA) models faces significant challenges in computational efficiency and data acquisition. We propose AcceRL, a fully asynchronous and decoupled RL framework designed to eliminate synchronization barriers by physically isolating training, inference, and rollouts. Crucially, AcceRL is the first to integrate a plug-and-play, trainable world model into a distributed asynchronous RL pipeline to generate virtual experiences. Experime...", "published": "2026-03-19", "arxiv_id": "2603.18464", "link": "https://arxiv.org/abs/2603.18464", "categories": [ "cs.LG" ] }, { "title": "Self-adapting Robotic Agents through Online Continual Reinforcement Learning with World Model Feedback", "authors": [ "Fabian Domberg", "Georg Schildbach" ], "summary": "As learning-based robotic controllers are typically trained offline and deployed with fixed parameters, their ability to cope with unforeseen changes during operation is limited. Biologically inspired, this work presents a framework for online Continual Reinforcement Learning that enables automated adaptation during deployment. Building on DreamerV3, a model-based Reinforcement Learning algorithm, the proposed method leverages world model prediction residuals to detect out-of-distribution events...", "published": "2026-03-04", "arxiv_id": "2603.04029", "link": "https://arxiv.org/abs/2603.04029", "categories": [ "cs.RO", "cs.AI" ] } ], "3d_scene": [ { "title": "HERMES++: Toward a Unified Driving World Model for 3D Scene Understanding and Generation", "authors": [ "Xin Zhou", "Dingkang Liang", "Xiwu Chen", "Feiyang Tan", "Dingyuan Zhang" ], "summary": "Driving world models serve as a pivotal technology for autonomous driving by simulating environmental dynamics. However, existing approaches predominantly focus on future scene generation, often overlooking comprehensive 3D scene understanding. Conversely, while Large Language Models (LLMs) demonstrate impressive reasoning capabilities, they lack the capacity to predict future geometric evolution, creating a significant disparity between semantic interpretation and physical simulation. To bridge...", "published": "2026-04-30", "arxiv_id": "2604.28196", "link": "https://arxiv.org/abs/2604.28196", "categories": [ "cs.CV" ] }, { "title": "3D-Anchored Lookahead Planning for Persistent Robotic Scene Memory via World-Model-Based MCTS", "authors": [ "Bronislav Sidik", "Dror Mizrahi" ], "summary": "We present 3D-Anchored Lookahead Planning (3D-ALP), a System 2 reasoning engine for robotic manipulation that combines Monte Carlo Tree Search (MCTS) with a 3D-consistent world model as the rollout oracle. Unlike reactive policies that evaluate actions from the current camera frame only, 3D-ALP maintains a persistent camera-to-world (c2w) anchor that survives occlusion, enabling accurate replanning to object positions that are no longer directly observable. On a 5-step sequential reach task requ...", "published": "2026-04-13", "arxiv_id": "2604.11302", "link": "https://arxiv.org/abs/2604.11302", "categories": [ "cs.RO", "cs.AI" ] }, { "title": "GaussianDWM: 3D Gaussian Driving World Model for Unified Scene Understanding and Multi-Modal Generation", "authors": [ "Tianchen Deng", "Xuefeng Chen", "Yi Chen", "Qu Chen", "Yuyao Xu" ], "summary": "Driving World Models (DWMs) have been developing rapidly with the advances of generative models. However, existing DWMs lack 3D scene understanding capabilities and can only generate content conditioned on input data, without the ability to interpret or reason about the driving environment. Moreover, current approaches represent 3D spatial information with point cloud or BEV features do not accurately align textual information with the underlying 3D scene. To address these limitations, we propos...", "published": "2025-12-29", "arxiv_id": "2512.23180", "link": "https://arxiv.org/abs/2512.23180", "categories": [ "cs.CV" ] }, { "title": "HERMES: A Unified Self-Driving World Model for Simultaneous 3D Scene Understanding and Generation", "authors": [ "Xin Zhou", "Dingkang Liang", "Sifan Tu", "Xiwu Chen", "Yikang Ding" ], "summary": "Driving World Models (DWMs) have become essential for autonomous driving by enabling future scene prediction. However, existing DWMs are limited to scene generation and fail to incorporate scene understanding, which involves interpreting and reasoning about the driving environment. In this paper, we present a unified Driving World Model named HERMES. We seamlessly integrate 3D scene understanding and future scene evolution (generation) through a unified framework in driving scenarios. Specifical...", "published": "2025-01-24", "arxiv_id": "2501.14729", "link": "https://arxiv.org/abs/2501.14729", "categories": [ "cs.CV" ] }, { "title": "InfiniCube: Unbounded and Controllable Dynamic 3D Driving Scene Generation with World-Guided Video Models", "authors": [ "Yifan Lu", "Xuanchi Ren", "Jiawei Yang", "Tianchang Shen", "Zhangjie Wu" ], "summary": "We present InfiniCube, a scalable method for generating unbounded dynamic 3D driving scenes with high fidelity and controllability. Previous methods for scene generation either suffer from limited scales or lack geometric and appearance consistency along generated sequences. In contrast, we leverage the recent advancements in scalable 3D representation and video models to achieve large dynamic scene generation that allows flexible controls through HD maps, vehicle bounding boxes, and text descri...", "published": "2024-12-05", "arxiv_id": "2412.03934", "link": "https://arxiv.org/abs/2412.03934", "categories": [ "cs.CV", "cs.AI", "cs.GR" ] }, { "title": "OpenSU3D: Open World 3D Scene Understanding using Foundation Models", "authors": [ "Rafay Mohiuddin", "Sai Manoj Prakhya", "Fiona Collins", "Ziyuan Liu", "André Borrmann" ], "summary": "In this paper, we present a novel, scalable approach for constructing open set, instance-level 3D scene representations, advancing open world understanding of 3D environments. Existing methods require pre-constructed 3D scenes and face scalability issues due to per-point feature vector learning, limiting their efficacy with complex queries. Our method overcomes these limitations by incrementally building instance-level 3D scene representations using 2D foundation models, efficiently aggregating ...", "published": "2024-07-19", "arxiv_id": "2407.14279", "link": "https://arxiv.org/abs/2407.14279", "categories": [ "cs.CV" ] }, { "title": "Scaling Diffusion Models to Real-World 3D LiDAR Scene Completion", "authors": [ "Lucas Nunes", "Rodrigo Marcuzzi", "Benedikt Mersch", "Jens Behley", "Cyrill Stachniss" ], "summary": "Computer vision techniques play a central role in the perception stack of autonomous vehicles. Such methods are employed to perceive the vehicle surroundings given sensor data. 3D LiDAR sensors are commonly used to collect sparse 3D point clouds from the scene. However, compared to human perception, such systems struggle to deduce the unseen parts of the scene given those sparse point clouds. In this matter, the scene completion task aims at predicting the gaps in the LiDAR measurements to achie...", "published": "2024-03-20", "arxiv_id": "2403.13470", "link": "https://arxiv.org/abs/2403.13470", "categories": [ "cs.CV" ] }, { "title": "Semantic Abstraction: Open-World 3D Scene Understanding from 2D Vision-Language Models", "authors": [ "Huy Ha", "Shuran Song" ], "summary": "We study open-world 3D scene understanding, a family of tasks that require agents to reason about their 3D environment with an open-set vocabulary and out-of-domain visual inputs - a critical skill for robots to operate in the unstructured 3D world. Towards this end, we propose Semantic Abstraction (SemAbs), a framework that equips 2D Vision-Language Models (VLMs) with new 3D spatial capabilities, while maintaining their zero-shot robustness. We achieve this abstraction using relevancy maps extr...", "published": "2022-07-23", "arxiv_id": "2207.11514", "link": "https://arxiv.org/abs/2207.11514", "categories": [ "cs.CV", "cs.RO" ] } ] }