4d_2025.md
July 23, 2026 · View on GitHub
🎉 2025 Accepted Papers
| Year | Title | Venue | Paper | Code | Project Page |
|---|---|---|---|---|---|
| 2025 | Optimizing 4D Gaussians for Dynamic Scene Video from Single Landscape Images | ICLR 2025 | Link | Link | Link |
| 2025 | GS-DiT: Advancing Video Generation with Pseudo 4D Gaussian Fields through Efficient Dense 3D Point Tracking | CVPR 2025 | Link | Link | Link |
| 2025 | Stereo4D: Learning How Things Move in 3D from Internet Stereo Videos | CVPR 2025 Oral | Link | Link | Link |
| 2025 | Uni4D: Unifying Visual Foundation Models for 4D Modeling from a Single Video | CVPR 2025 Highlight | Link | Link | Link |
| 2025 | 4D-Fly: Fast 4D Reconstruction from a Single Monocular Video | CVPR 2025 | Link | Coming Soon! | Link |
| 2025 | GenMOJO: Robust Multi-Object 4D Generation for In-the-wild Videos | CVPR 2025 | Link | Link | Link |
| 2025 | Articulated Kinematics Distillation from Video Diffusion Models | CVPR 2025 | Link | -- | Link |
| 2025 | Free4D: Tuning-free 4D Scene Generation with Spatial-Temporal Consistency | ICCV 2025 | Link | Link | Link |
| 2025 | St4RTrack: Simultaneous 4D Reconstruction and Tracking in the World | ICCV 2025 | Link | Link | Link |
| 2025 | VLM4D: Towards Spatiotemporal Awareness in Vision Language Models | ICCV 2025 | Link | Link | Link |
| 2025 | Express4D: Expressive, Friendly, and Extensible 4D Facial Motion Generation Benchmark | ICCV DataCV Workshop 2025 | Link | Link | Link |
| 2025 | CityDreamer4D: Compositional Generative Model of Unbounded 4D Cities | TPAMI 2025 | Link | Link | Link |
| 2025 | TesserAct: Learning 4D Embodied World Models | ICCV 2025 | Link | Link | Link |
| 2025 | T2Bs: Text-to-Character Blendshapes via Video Generation | ICCV 2025 | Link | Link | Link |
| 2025 | Geo4D: Leveraging Video Generators for Geometric 4D Scene Reconstruction | ICCV 2025 Highlight | Link | Link | Link |
| 2025 | SV4D 2.0: Enhancing Spatio-Temporal Consistency in Multi-View Video Diffusion for High-Quality 4D Generation | ICCV 2025 | Link | Link | Link |
| 2025 | HoloTime: Taming Video Diffusion Models for Panoramic 4D Scene Generation | ACM MM 2025 | Link | Link | Link |
| 2025 | Stable Part Diffusion 4D: Multi-View RGB and Kinematic Parts Video Generation | NeurIPS 2025 | Link | Link | Link |
| 2025 | In-2-4D: Inbetweening from Two Single-View Images to 4D Generation | SIGGRAPH ASIA 2025 | Link | Link | Link |
| 2025 | Animus3D: Text-driven 3D Animation via Motion Score Distillation | SIGGRAPH ASIA 2025 | Link | Link | Link |
| 2025 | Track, Inpaint, Resplat: Subject-driven 3D and 4D Generation with Progressive Texture Infilling | NeurIPS 2025 | Link | Link | Link |
| 2025 | 4Real-Video-V2: Fused View-Time Attention and Feedforward Reconstruction for 4D Scene Generation | NeurIPS 2025 | Link | -- | Link |
| 2025 | 4D-LRM: Large Space-Time Reconstruction Model From and To Any View at Any Time | NeurIPS 2025 | Link | Link | Link |
| 2025 | TwoSquared: 4D Generation from 2D Image Pairs | 3DV 2026 Oral | Link | Link | Link |
| 2025 | OmniWorld: A Multi-Domain and Multi-Modal Dataset for 4D World Modeling | ICLR 2026 | Link | Link | Link |
| 2025 | Lyra: Generative 3D Scene Reconstruction via Video Diffusion Model Self-Distillation | ICLR 2026 | Link | Link | Link |
| 2025 | ShapeGen4D: Towards High Quality 4D Shape Generation from Videos | ICLR 2026 | Link | -- | Link |
| 2026 | One4D: Unified 4D Generation and Reconstruction via Decoupled LoRA Control | ECCV 2026 | Link | Link | Link |
| 2025 | AR4D: Autoregressive 4D Generation from Monocular Videos | CVPR 2026 | Link | -- | Link |
| 2025 | MoVieS: Motion-Aware 4D Dynamic View Synthesis in One Second | CVPR 2026 | Link | Link | Link |
| 2025 | Object-Aware 4D Human Motion Generation | CVPR 2026 | Link | -- | -- |
| 2025 | SeeU: Seeing the Unseen World via 4D Dynamics-aware Generation | CVPR 2026 | Link | Link | Link |
| 2025 | Sonic4D: Spatial Audio Generation for Immersive 4D Scene Exploration | AAAI 2026 | Link | Link | Link |
| 2025 | SEE4D: Pose-Free 4D Generation via Auto-Regressive Video Inpainting | Eurographics 2026 | Link | -- | Link |
| 2025 | Joint 3D Geometry Reconstruction and Motion Generation for 4D Synthesis from a Single Image | ECCV 2026 | Link | Link | Link |
Accepted Papers References
%accepted papers
@inproceedings{jinoptimizing,
title={Optimizing 4D Gaussians for Dynamic Scene Video from Single Landscape Images},
author={Jin, In-Hwan and Choo, Haesoo and Jeong, Seong-Hun and Heemoon, Park and Kim, Junghwan and Kwon, Oh-joon and Kong, Kyeongbo},
booktitle={The Thirteenth International Conference on Learning Representations},
year={2025}
}
@article{bian2025gsdit,
title={GS-DiT: Advancing Video Generation with Pseudo 4D Gaussian Fields through Efficient Dense 3D Point Tracking},
author={Bian, Weikang and Huang, Zhaoyang and Shi, Xiaoyu and and Li, Yijin and Wang, Fu-Yun and Li, Hongsheng},
journal={arXiv preprint arXiv:2501.02690},
year={2025}
}
@article{jin2024stereo4d,
title={Stereo4D: Learning How Things Move in 3D from Internet Stereo Videos},
author={Jin, Linyi and Tucker, Richard and Li, Zhengqi and Fouhey, David and Snavely, Noah and Holynski, Aleksander},
journal={CVPR},
year={2025},
}
@article{yao2025uni4d,
title={Uni4D: Unifying Visual Foundation Models for 4D Modeling from a Single Video},
author={Yao, David Yifan and Zhai, Albert J and Wang, Shenlong},
journal={arXiv preprint arXiv:2503.21761},
year={2025}
}
@inproceedings{wu20254d,
title={4D-Fly: Fast 4D Reconstruction from a Single Monocular Video},
author={Wu, Diankun and Liu, Fangfu and Hung, Yi-Hsin and Qian, Yue and Zhan, Xiaohang and Duan, Yueqi},
booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference},
pages={16663--16673},
year={2025}
}
@inproceedings{chu2025robust,
title={Robust Multi-Object 4D Generation for In-the-wild Videos},
author={Chu, Wen-Hsuan and Ke, Lei and Liu, Jianmeng and Huo, Mingxiao and Tokmakov, Pavel and Fragkiadaki, Katerina},
booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference},
pages={22067--22077},
year={2025}
}
@inproceedings{li2025articulated,
title={Articulated Kinematics Distillation from Video Diffusion Models},
author={Li, Xuan and Ma, Qianli and Lin, Tsung-Yi and Chen, Yongxin and Jiang, Chenfanfu and Liu, Ming-Yu and Xiang, Donglai},
booktitle={Proceedings of the Computer Vision and Pattern Recognition Conference},
pages={17571--17581},
year={2025}
}
@article{liu2025free4d,
title={Free4D: Tuning-free 4D Scene Generation with Spatial-Temporal Consistency},
author={Liu, Tianqi and Huang, Zihao and Chen, Zhaoxi and Wang, Guangcong and Hu, Shoukang and Shen, liao and Sun, Huiqiang and Cao, Zhiguo and Li, Wei and Liu, Ziwei},
journal={arXiv preprint arXiv:2503.20785},
year={2025}
}
@inproceedings{st4rtrack2025,
title={St4RTrack: Simultaneous 4D Reconstruction and Tracking in the World},
author={Feng*, Haiwen and Zhang*, Junyi and Wang, Qianqian and Ye, Yufei and Yu, Pengcheng and Black, Michael J. and Darrell, Trevor and Kanazawa, Angjoo},
booktitle={Proceedings of the IEEE/CVF International Conference on Computer Vision},
year={2025}
}
@inproceedings{zhou2025vlm4d,
title={VLM4D: Towards Spatiotemporal Awareness in Vision Language Models},
author={Zhou, Shijie and Vilesov, Alexander and He, Xuehai and Wan, Ziyu and Zhang, Shuwang and Nagachandra, Aditya and Chang, Di and Chen, Dongdong and Wang, Eric Xin and Kadambi, Achuta},
booktitle={Proceedings of the IEEE/CVF international conference on computer vision},
year={2025}
}
@misc{aloni2025express4dexpressivefriendlyextensible,
title={Express4D: Expressive, Friendly, and Extensible 4D Facial Motion Generation Benchmark},
author={Yaron Aloni and Rotem Shalev-Arkushin and Yonatan Shafir and Guy Tevet and Ohad Fried and Amit Haim Bermano},
year={2025},
eprint={2508.12438},
archivePrefix={arXiv},
primaryClass={cs.GR},
url={https://arxiv.org/abs/2508.12438}
}
@article{xie2025citydreamer4d,
title={CityDreamer4D: Compositional generative model of unbounded 4D cities},
author={Xie, Haozhe and Chen, Zhaoxi and Hong, Fangzhou and Liu, Ziwei},
journal={arXiv e-prints},
pages={arXiv--2501},
year={2025}
}
@article{zhen2025tesseract,
title={TesserAct: learning 4D embodied world models},
author={Zhen, Haoyu and Sun, Qiao and Zhang, Hongxin and Li, Junyan and Zhou, Siyuan and Du, Yilun and Gan, Chuang},
journal={arXiv preprint arXiv:2504.20995},
year={2025}
}
@misc{luo2025t2bstexttocharacterblendshapesvideo,
title={T2Bs: Text-to-Character Blendshapes via Video Generation},
author={Jiahao Luo and Chaoyang Wang and Michael Vasilkovsky and Vladislav Shakhrai and Di Liu and Peiye Zhuang and Sergey Tulyakov and Peter Wonka and Hsin-Ying Lee and James Davis and Jian Wang},
year={2025},
eprint={2509.10678},
archivePrefix={arXiv},
primaryClass={cs.GR},
url={https://arxiv.org/abs/2509.10678},
}
@article{jiang2025geo4d,
title={Geo4d: Leveraging video generators for geometric 4d scene reconstruction},
author={Jiang, Zeren and Zheng, Chuanxia and Laina, Iro and Larlus, Diane and Vedaldi, Andrea},
journal={arXiv preprint arXiv:2504.07961},
year={2025}
}
@article{yao2025sv4d,
title={Sv4d 2.0: Enhancing spatio-temporal consistency in multi-view video diffusion for high-quality 4d generation},
author={Yao, Chun-Han and Xie, Yiming and Voleti, Vikram and Jiang, Huaizu and Jampani, Varun},
journal={arXiv preprint arXiv:2503.16396},
year={2025}
}
@article{zhou2025holotime,
title={HoloTime: Taming Video Diffusion Models for Panoramic 4D Scene Generation},
author={Zhou, Haiyang and Yu, Wangbo and Guan, Jiawen and Cheng, Xinhua and Tian, Yonghong and Yuan, Li},
journal={arXiv preprint arXiv:2504.21650},
year={2025}
}
@article{zhang2025stable,
title={Stable Part Diffusion 4D: Multi-View RGB and Kinematic Parts Video Generation},
author={Zhang, Hao and Yao, Chun-Han and Donn{\'e}, Simon and Ahuja, Narendra and Jampani, Varun},
journal={arXiv preprint arXiv:2509.10687},
year={2025}
}
@article{nag20252,
title={In-2-4d: Inbetweening from two single-view images to 4d generation},
author={Nag, Sauradip and Cohen-Or, Daniel and Zhang, Hao and Mahdavi-Amiri, Ali},
journal={arXiv preprint arXiv:2504.08366},
year={2025}
}
@inproceedings{sun2025animus3d,
title={Animus3D: Text-driven 3D Animation via Motion Score Distillation},
author={Sun, Qi and Wang, Can and Shang, Jiaxiang and Feng, Wensen and Liao, Jing},
booktitle={Proceedings of the SIGGRAPH Asia 2025 Conference Papers},
pages={1--11},
year={2025}
}
@inproceedings{zheng2025trackinpaintresplat,
title={Track, Inpaint, Resplat: Subject-driven 3D and 4D Generation with Progressive Texture Infilling},
author={Zheng, Shuhong and Mirzaei, Ashkan and Gilitschenski, Igor},
booktitle={NeurIPS},
year={2025}
}
@article{wang20254real,
title={4Real-Video-V2: Fused View-Time Attention and Feedforward Reconstruction for 4D Scene Generation},
author={Wang, Chaoyang and Mirzaei, Ashkan and Goel, Vidit and Menapace, Willi and Siarohin, Aliaksandr and Vinella, Avalon and Vasilkovsky, Michael and Skorokhodov, Ivan and Shakhrai, Vladislav and Korolev, Sergey and others},
journal={arXiv preprint arXiv:2506.18839},
year={2025}
}
@article{ma20254d,
title={4D-LRM: Large Space-Time Reconstruction Model From and To Any View at Any Time},
author={Ma, Ziqiao and Chen, Xuweiyi and Yu, Shoubin and Bi, Sai and Zhang, Kai and Ziwen, Chen and Xu, Sihan and Yang, Jianing and Xu, Zexiang and Sunkavalli, Kalyan and others},
journal={arXiv preprint arXiv:2506.18890},
year={2025}
}
@inproceedings{sang2025twosquared,
title={Twosquared: 4d generation from 2d image pairs},
author={Sang, Lu and Canfes, Zehranaz and Marin, Riccardo and Cao, Dongliang and Bernard, Florian and Cremers, Daniel},
booktitle={Thirteenth International Conference on 3D Vision},
year={2025}
}
@article{zhou2025omniworld,
title={Omniworld: A multi-domain and multi-modal dataset for 4d world modeling},
author={Zhou, Yang and Wang, Yifan and Zhou, Jianjun and Chang, Wenzheng and Guo, Haoyu and Li, Zizun and Ma, Kaijing and Li, Xinyue and Wang, Yating and Zhu, Haoyi and others},
journal={arXiv preprint arXiv:2509.12201},
year={2025}
}
@article{bahmani2025lyra,
title={Lyra: Generative 3D Scene Reconstruction via Video Diffusion Model Self-Distillation},
author={Bahmani, Sherwin and Shen, Tianchang and Ren, Jiawei and Huang, Jiahui and Jiang, Yifeng and Turki, Haithem and Tagliasacchi, Andrea and Lindell, David B and Gojcic, Zan and Fidler, Sanja and others},
journal={arXiv preprint arXiv:2509.19296},
year={2025}
}
@article{yenphraphai2025shapegen4d,
title={Shapegen4d: Towards high quality 4d shape generation from videos},
author={Yenphraphai, Jiraphon and Mirzaei, Ashkan and Chen, Jianqi and Zou, Jiaxu and Tulyakov, Sergey and Yeh, Raymond A and Wonka, Peter and Wang, Chaoyang},
journal={arXiv preprint arXiv:2510.06208},
year={2025}
}
@article{mi2025one4d,
title={One4D: Unified 4D Generation and Reconstruction via Decoupled LoRA Control},
author={Mi, Zhenxing and Wang, Yuxin and Xu, Dan},
journal={arXiv preprint arXiv:2511.18922},
year={2025}
}
@inproceedings{zhu2026ar4d,
title={Ar4d: Autoregressive 4d generation from monocular videos},
author={Zhu, Hanxin and He, Tianyu and Chen, Zhibo},
booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
pages={88--98},
year={2026}
}
@inproceedings{lin2026movies,
title={Movies: Motion-aware 4d dynamic view synthesis in one second},
author={Lin, Chenguo and Lin, Yuchen and Pan, Panwang and Yu, Yifan and Hu, Tao and Yan, Honglei and Fragkiadaki, Katerina and Mu, Yadong},
booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
pages={295--306},
year={2026}
}
@inproceedings{gui2026object,
title={Object-Aware 4D Human Motion Generation},
author={Gui, Shurui and Patel, Deep and Li, Xiner and Min, Martin Renqiang},
booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
pages={11563--11572},
year={2026}
}
@inproceedings{yuan2026seeu,
title={SeeU: Seeing the Unseen World via 4D Dynamics-aware Generation},
author={Yuan, Yu and Wickremasinghe, Tharindu and Nadir, Zeeshan and Wang, Xijun and Chi, Yiheng and Chan, Stanley H},
booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
pages={11150--11162},
year={2026}
}
@inproceedings{xie2026sonic4d,
title={Sonic4d: Spatial audio generation for immersive 4d scene exploration},
author={Xie, Siyi and Zhu, Hanxin and Chen, Xinyi and He, Tianyu and Li, Xin and Chen, Zhibo},
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
volume={40},
number={13},
pages={11087--11095},
year={2026}
}
@inproceedings{lu2026see4d,
title={See4D: Pose-Free 4D Generation via Auto-Regressive Video Inpainting},
author={Lu, Dongyue and Liang, Ao and Huang, Tianxin and Fu, Xiao and Zhao, Yuyang and Ma, Baorui and Pan, Liang and Yin, Wei and Kong, Lingdong and Ooi, Wei Tsang and others},
booktitle={Computer Graphics Forum},
pages={e70345},
year={2026},
organization={Wiley Online Library}
}
@article{zhang2025joint,
title={Joint 3D Geometry Reconstruction and Motion Generation for 4D Synthesis from a Single Image},
author={Zhang, Yanran and Wang, Ziyi and Zheng, Wenzhao and Zhu, Zheng and Zhou, Jie and Lu, Jiwen},
journal={arXiv preprint arXiv:2512.05044},
year={2025}
}
💡 2025 ArXiv Papers
1. WideRange4D: Enabling High-Quality 4D Reconstruction with Wide-Range Movements and Scenes
Ling Yang, Kaixin Zhu, Juanxi Tian, Bohan Zeng, Mingbao Lin, Hongjuan Pei, Wentao Zhang, Shuicheng Yan
(Peking University, University of the Chinese Academy of Sciences, National University of Singapore)
Abstract
With the rapid development of 3D reconstruction technology, research in 4D reconstruction is also advancing, existing 4D reconstruction methods can generate high-quality 4D scenes. However, due to the challenges in acquiring multi-view video data, the current 4D reconstruction benchmarks mainly display actions performed in place, such as dancing, within limited scenarios. In practical scenarios, many scenes involve wide-range spatial movements, highlighting the limitations of existing 4D reconstruction datasets. Additionally, existing 4D reconstruction methods rely on deformation fields to estimate the dynamics of 3D objects, but deformation fields struggle with wide-range spatial movements, which limits the ability to achieve high-quality 4D scene reconstruction with wide-range spatial movements. In this paper, we focus on 4D scene reconstruction with significant object spatial movements and propose a novel 4D reconstruction benchmark, WideRange4D. This benchmark includes rich 4D scene data with large spatial variations, allowing for a more comprehensive evaluation of the generation capabilities of 4D generation methods. Furthermore, we introduce a new 4D reconstruction method, Progress4D, which generates stable and high-quality 4D results across various complex 4D scene reconstruction tasks. We conduct both quantitative and qualitative comparison experiments on WideRange4D, showing that our Progress4D outperforms existing state-of-the-art 4D reconstruction methods.2. DeepVerse: 4D Autoregressive Video Generation as a World Model
Junyi Chen, Haoyi Zhu, Xianglong He, Yifan Wang, Jianjun Zhou, Wenzheng Chang, Yang Zhou, Zizun Li, Zhoujie Fu, Jiangmiao Pang, Tong He
(Shanghai Jiao Tong University, Shanghai AI Lab, University of Science and Technology of China, Tsinghua University, Zhejiang University, Fudan University, Nanyang Technology University)
Abstract
World models serve as essential building blocks toward Artificial General Intelligence (AGI), enabling intelligent agents to predict future states and plan actions by simulating complex physical interactions. However, existing interactive models primarily predict visual observations, thereby neglecting crucial hidden states like geometric structures and spatial coherence. This leads to rapid error accumulation and temporal inconsistency. To address these limitations, we introduce DeepVerse, a novel 4D interactive world model explicitly incorporating geometric predictions from previous timesteps into current predictions conditioned on actions. Experiments demonstrate that by incorporating explicit geometric constraints, DeepVerse captures richer spatio-temporal relationships and underlying physical dynamics. This capability significantly reduces drift and enhances temporal consistency, enabling the model to reliably generate extended future sequences and achieve substantial improvements in prediction accuracy, visual realism, and scene rationality. Furthermore, our method provides an effective solution for geometry-aware memory retrieval, effectively preserving long-term spatial consistency. We validate the effectiveness of DeepVerse across diverse scenarios, establishing its capacity for high-fidelity, long-horizon predictions grounded in geometry-aware dynamics.3. BulletGen: Improving 4D Reconstruction with Bullet-Time Generation
Denys Rozumnyi, Jonathon Luiten, Numair Khan, Johannes Schönberger, Peter Kontschieder (Meta Reality Labs)
Abstract
Transforming casually captured, monocular videos into fully immersive dynamic experiences is a highly ill-posed task, and comes with significant challenges, e.g., reconstructing unseen regions, and dealing with the ambiguity in monocular depth estimation. In this work we introduce BulletGen, an approach that takes advantage of generative models to correct errors and complete missing information in a Gaussian-based dynamic scene representation. This is done by aligning the output of a diffusion-based video generation model with the 4D reconstruction at a single frozen "bullet-time" step. The generated frames are then used to supervise the optimization of the 4D Gaussian model. Our method seamlessly blends generative content with both static and dynamic scene components, achieving state-of-the-art results on both novel-view synthesis, and 2D/3D tracking tasks.4. 4DNeX: Feed-Forward 4D Generative Modeling Made Easy
Zhaoxi Chen, Tianqi Liu, Long Zhuo, Jiawei Ren, Zeng Tao, He Zhu, Fangzhou Hong, Liang Pan, Ziwei Liu
(Nanyang Technological University, Shanghai AI Laboratory)
Abstract
We present 4DNeX, the first feed-forward framework for generating 4D (i.e., dynamic 3D) scene representations from a single image. In contrast to existing methods that rely on computationally intensive optimization or require multi-frame video inputs, 4DNeX enables efficient, end-to-end image-to-4D generation by fine-tuning a pretrained video diffusion model. Specifically, 1) to alleviate the scarcity of 4D data, we construct 4DNeX-10M, a large-scale dataset with high-quality 4D annotations generated using advanced reconstruction approaches. 2) we introduce a unified 6D video representation that jointly models RGB and XYZ sequences, facilitating structured learning of both appearance and geometry. 3) we propose a set of simple yet effective adaptation strategies to repurpose pretrained video diffusion models for 4D modeling. 4DNeX produces high-quality dynamic point clouds that enable novel-view video synthesis. Extensive experiments demonstrate that 4DNeX outperforms existing 4D generation methods in efficiency and generalizability, offering a scalable solution for image-to-4D modeling and laying the foundation for generative 4D world models that simulate dynamic scene evolution.5. Diff4Splat: Controllable 4D Scene Generation with Latent Dynamic Reconstruction Models
Panwang Pan, Chenguo Lin, Jingjing Zhao, Chenxin Li, Yuchen Lin, Haopeng Li, Honglei Yan, Kairun Wen, Yunlong Lin, Yixuan Yuan, Yadong Mu
(Peking University, The Chinese University of Hong Kong, Xiamen University)
Abstract
We introduce Diff4Splat, a feed-forward method that synthesizes controllable and explicit 4D scenes from a single image. Our approach unifies the generative priors of video diffusion models with geometry and motion constraints learned from large-scale 4D datasets. Given a single input image, a camera trajectory, and an optional text prompt, Diff4Splat directly predicts a deformable 3D Gaussian field that encodes appearance, geometry, and motion, all in a single forward pass, without test-time optimization or post-hoc refinement. At the core of our framework lies a video latent transformer, which augments video diffusion models to jointly capture spatio-temporal dependencies and predict time-varying 3D Gaussian primitives. Training is guided by objectives on appearance fidelity, geometric accuracy, and motion consistency, enabling Diff4Splat to synthesize high-quality 4D scenes in 30 seconds. We demonstrate the effectiveness of Diff4Splatacross video generation, novel view synthesis, and geometry extraction, where it matches or surpasses optimization-based methods for dynamic scene synthesis while being significantly more efficient.6. SWiT-4D: Sliding-Window Transformer for Lossless and Parameter-Free Temporal 4D Generation
Kehong Gong, Zhengyu Wen, Mingxi Xu, Weixia He, Qi Wang, Ning Zhang, Zhengyu Li, Chenbin Li, Dongze Lian, Wei Zhao, Xiaoyu He, Mingyuan Zhang
(Huawei Technologies Co., Ltd., Huawei Central Media Technology Institute)
Abstract
Despite significant progress in 4D content generation, the conversion of monocular videos into high-quality animated 3D assets with explicit 4D meshes remains considerably challenging. The scarcity of large-scale, naturally captured 4D mesh datasets further limits the ability to train generalizable video-to-4D models from scratch in a purely data-driven manner. Meanwhile, advances in image-to-3D generation, supported by extensive datasets, offer powerful prior models that can be leveraged. To better utilize these priors while minimizing reliance on 4D supervision, we introduce SWiT-4D, a Sliding-Window Transformer for lossless, parameter-free temporal 4D mesh generation. SWiT-4D integrates seamlessly with any Diffusion Transformer (DiT)-based image-to-3D generator, adding spatial-temporal modeling across video frames while preserving the original single-image forward process, enabling 4D mesh reconstruction from videos of arbitrary length. To recover global translation, we further introduce an optimization-based trajectory module tailored for static-camera monocular videos. SWiT-4D demonstrates strong data efficiency: with only a single short (<10s) video for fine-tuning, it achieves high-fidelity geometry and stable temporal consistency, indicating practical deployability under extremely limited 4D supervision. Comprehensive experiments on both in-domain zoo-test sets and challenging out-of-domain benchmarks (C4D, Objaverse, and in-the-wild videos) show that SWiT-4D consistently outperforms existing baselines in temporal smoothness.| Year | Title | ArXiv Time | Paper | Code | Project Page |
|---|---|---|---|---|---|
| 2025 | WideRange4D: Enabling High-Quality 4D Reconstruction with Wide-Range Movements and Scenes | 17 Mar 2025 | Link | Link | Dataset Page |
| 2025 | DeepVerse: 4D Autoregressive Video Generation as a World Model | 1 Jun 2025 | Link | Link | Link |
| 2025 | BulletGen: Improving 4D Reconstruction with Bullet-Time Generation | 23 Jun 2025 | Link | -- | -- |
| 2025 | 4DNeX: Feed-Forward 4D Generative Modeling Made Easy | 18 Aug 2025 | Link | Link | Link |
| 2025 | Diff4Splat: Controllable 4D Scene Generation with Latent Dynamic Reconstruction Models | 1 Nov 2025 | Link | -- | Link |
| 2025 | SWiT-4D: Sliding-Window Transformer for Lossless and Parameter-Free Temporal 4D Generation | 11 Dec 2025 | Link | -- | Link |
ArXiv Papers References
%axiv papers
@article{yang2025widerange4d,
title={WideRange4D: Enabling High-Quality 4D Reconstruction with Wide-Range Movements and Scenes},
author={Yang, Ling and Zhu, Kaixin and Tian, Juanxi and Zeng, Bohan and Lin, Mingbao and Pei, Hongjuan and Zhang, Wentao and Yan, Shuichen},
journal={arXiv preprint arXiv:2503.13435},
year={2025}
}
@misc{chen2025deepverse4dautoregressivevideo,
title={DeepVerse: 4D Autoregressive Video Generation as a World Model},
author={Junyi Chen and Haoyi Zhu and Xianglong He and Yifan Wang and Jianjun Zhou and Wenzheng Chang and Yang Zhou and Zizun Li and Zhoujie Fu and Jiangmiao Pang and Tong He},
year={2025},
eprint={2506.01103},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2506.01103},
}
@misc{rozumnyi2025bulletgenimproving4dreconstruction,
title={BulletGen: Improving 4D Reconstruction with Bullet-Time Generation},
author={Denys Rozumnyi and Jonathon Luiten and Numair Khan and Johannes Schönberger and Peter Kontschieder},
year={2025},
eprint={2506.18601},
archivePrefix={arXiv},
primaryClass={cs.GR},
url={https://arxiv.org/abs/2506.18601},
}
@article{chen20254dnex,
title={4DNeX: Feed-Forward 4D Generative Modeling Made Easy},
author={Chen, Zhaoxi and Liu, Tianqi and Zhuo, Long and Ren, Jiawei and Tao, Zeng and Zhu, He and Hong, Fangzhou and Pan, Liang and Liu, Ziwei},
journal={arXiv preprint arXiv:2508.13154},
year={2025}
}
@article{pan2025diff4splat,
title={Diff4Splat: Controllable 4D Scene Generation with Latent Dynamic Reconstruction Models},
author={Pan, Panwang and Lin, Chenguo and Zhao, Jingjing and Li, Chenxin and Lin, Yuchen and Li, Haopeng and Yan, Honglei and Wen, Kairun and Lin, Yunlong and Yuan, Yixuan and others},
journal={arXiv preprint arXiv:2511.00503},
year={2025}
}
@article{gong2025swit4d,
title = {SWiT-4D: Sliding-Window Transformer for Lossless and Parameter-Free Temporal 4D Generation},
author = {Gong, Kehong and Wen, Zhengyu and Xu, Mingxi and He, Weixia and Wang, Qi and
Zhang, Ning and Li, Zhengyu and Li, Chenbin and Lian, Dongze and
Zhao, Wei and He, Xiaoyu and Zhang, Mingyuan},
journal = {arXiv preprint arXiv:2512.10860},
year = {2025}
}
2025 arXiv Survey
- [11 Sep 2025]3D and 4D World Modeling: A Survey [Paper][GitHub]
- [22 Oct 2025]Advances in 4D Representation: Geometry, Motion, and Interaction [Paper][Project Page]