4d_2024.md

July 23, 2026 ยท View on GitHub

๐ŸŽ‰ 2024 Accepted

YearTitleVenuePaperCodeProject Page
2024TC4D: Trajectory-Conditioned Text-to-4D GenerationECCV 2024LinkLinkLink
2024STAG4D: Spatial-Temporal Anchored Generative 4D GaussiansECCV 2024LinkLinkLink
2024SC4D: Sparse-Controlled Video-to-4D Generation and Motion TransferECCV 2024LinkLinkLink
2024DreamScene4D: Dynamic Multi-Object Scene Generation from Monocular VideosNeurIPS 2024LinkLinkLink
20244Diffusion: Multi-view Video Diffusion Model for 4D GenerationNeurIPS 2024LinkLinkLink
2024DreamMesh4D: Video-to-4D Generation with Sparse-Controlled Gaussian-Mesh Hybrid RepresentationNeurIPS 2024LinkLinkLink
2024L4GM: Large 4D Gaussian Reconstruction ModelNeurIPS 2024LinkLinkLink
20244Real: Towards Photorealistic 4D Scene Generation via Video Diffusion ModelsNeurIPS 2024Link--Link
2024Animate3D: Animating Any 3D Model with Multi-view Video DiffusionNeurIPS 2024LinkLinkLink
2024Compositional 3D-aware Video Generation with LLM DirectorNeurIPS 2024Link--Link
2024Vidu4D: Single Generated Video to High-Fidelity 4D Reconstruction with Dynamic Gaussian SurfelsNeurIPS 2024LinkLinkLink
2024Diffusion4D: Fast Spatial-temporal Consistent 4D Generation via Video Diffusion ModelsNeurIPS 2024LinkLinkLink
2024MonST3R: A Simple Approach for Estimating Geometry in the Presence of MotionICLR 2025 (Spotlight)LinkLinkLink
20244K4DGen: Panoramic 4D Generation at 4K ResolutionICLR 2025LinkLinkLink
2024GenXD: Generating Any 3D and 4D ScenesICLR 2025LinkLinkLink
2024AvatarGO: Zero-shot 4D Human-Object Interaction Generation and AnimationICLR 2025LinkLinkLink
2024EG4D: Explicit Generation of 4D Object without Score DistillationICLR 2025LinkLink--
20244-LEGS: 4D Language Embedded Gaussian SplattingEurographics 2025LinkLinkLink
2024Disco4D: Disentangled 4D Human Generation and Animation from a Single ImageCVPR 2025LinkLinkLink
2024CAT4D: Create Anything in 4D with Multi-View Video Diffusion ModelsCVPR 2025Link--Link
20244Real-Video: Learning Generalizable Photo-Realistic 4D Video DiffusionCVPR 2025Link--Link
2024STAR: Skeleton-aware Text-based 4D Avatar Generation with In-Network Motion RetargetingTVCGLinkLinkLink
2024Shape of Motion: 4D Reconstruction from a Single VideoICCV 2025 (Highlight)LinkLinkLink
2024DimensionX: Create Any 3D and 4D Scenes from a Single Image with Controllable Video DiffusionICCV 2025LinkLinkLink
2024Fast Dynamic 3D Object Generation from a Single-view VideoIJCV 2026LinkLinkLink
2024Comp4D: LLM-Guided Compositional 4D Scene GenerationWACV 2026LinkLinkLink
2024ZeroHSI: Zero-Shot 4D Human-Scene Interaction by Video Generation3DV 2026Link--Link
Accepted Papers References
%accepted papers

%text to 4d (ECCV24)
@inproceedings{bahmani2025tc4d,
  title={Tc4d: Trajectory-conditioned text-to-4d generation},
  author={Bahmani, Sherwin and Liu, Xian and Yifan, Wang and Skorokhodov, Ivan and Rong, Victor and Liu, Ziwei and Liu, Xihui and Park, Jeong Joon and Tulyakov, Sergey and Wetzstein, Gordon and others},
  booktitle={European Conference on Computer Vision},
  pages={53--72},
  year={2025},
  organization={Springer}
}

%text/Image and video to 4d (ECCV24)
@inproceedings{zeng2025stag4d,
  title={Stag4d: Spatial-temporal anchored generative 4d gaussians},
  author={Zeng, Yifei and Jiang, Yanqin and Zhu, Siyu and Lu, Yuanxun and Lin, Youtian and Zhu, Hao and Hu, Weiming and Cao, Xun and Yao, Yao},
  booktitle={European Conference on Computer Vision},
  pages={163--179},
  year={2025},
  organization={Springer}
}

%video to 4d (ECCV24 and NIPS24)
@inproceedings{wu2025sc4d,
  title={Sc4d: Sparse-controlled video-to-4d generation and motion transfer},
  author={Wu, Zijie and Yu, Chaohui and Jiang, Yanqin and Cao, Chenjie and Wang, Fan and Bai, Xiang},
  booktitle={European Conference on Computer Vision},
  pages={361--379},
  year={2025},
  organization={Springer}
}

@inproceedings{dreamscene4d,
  title={DreamScene4D: Dynamic Multi-Object Scene Generation from Monocular Videos},
  author={Chu, Wen-Hsuan and Ke, Lei and Fragkiadaki, Katerina},
  booktitle={NeurIPS},
  year={2024}
}

@article{zhang20244diffusion,
    title={4Diffusion: Multi-view Video Diffusion Model for 4D Generation}, 
    author={Haiyu Zhang and Xinyuan Chen and Yaohui Wang and Xihui Liu and Yunhong Wang and Yu Qiao},
    year={2024}
}

@inproceedings{li2024dreammesh4d,
    title={DreamMesh4D: Video-to-4D Generation with Sparse-Controlled Gaussian-Mesh Hybrid Representation},
    author={Zhiqi Li and Yiming Chen and Peidong Liu},
    booktitle={Advances in Neural Information Processing Systems (NeurIPS)},
    year={2024}
}

@article{ren2024l4gm,
    title={L4GM: Large 4D Gaussian Reconstruction Model},
    author={Ren, Jiawei and Xie, Kevin and Mirzaei, Ashkan and Liang, Hanxue and Zeng, Xiaohui and Kreis, Karsten and Liu, Ziwei and Torralba, Antonio and Fidler, Sanja and Kim, Seung Wook and Ling, Huan},
    title={arXiv preprint arXiv:2406.xxxxx},
    year={2024}
}

%text to 4d (NIPS24)
@article{yu20244real,
  title={4Real: Towards Photorealistic 4D Scene Generation via Video Diffusion Models},
  author={Yu, Heng and Wang, Chaoyang and Zhuang, Peiye and Menapace, Willi and Siarohin, Aliaksandr and Cao, Junli and Jeni, Laszlo A and Tulyakov, Sergey and Lee, Hsin-Ying},
  journal={arXiv preprint arXiv:2406.07472},
  year={2024}
}

@article{jiang2024animate3d,
  title={Animate3d: Animating any 3d model with multi-view video diffusion},
  author={Jiang, Yanqin and Yu, Chaohui and Cao, Chenjie and Wang, Fan and Hu, Weiming and Gao, Jin},
  journal={arXiv preprint arXiv:2407.11398},
  year={2024}
}

@article{zhu2024compositional,
  title={Compositional 3d-aware video generation with llm director},
  author={Zhu, Hanxin and He, Tianyu and Tang, Anni and Guo, Junliang and Chen, Zhibo and Bian, Jiang},
  journal={arXiv preprint arXiv:2409.00558},
  year={2024}
}

@article{wang2024vidu4d,
  title={Vidu4D: Single Generated Video to High-Fidelity 4D Reconstruction with Dynamic Gaussian Surfels},
  author={Wang, Yikai and Wang, Xinzhou and Chen, Zilong and Wang, Zhengyi and Sun, Fuchun and Zhu, Jun},
  journal={arXiv preprint arXiv:2405.16822},
  year={2024}
}

@article{liang2024diffusion4d,
  title={Diffusion4D: Fast Spatial-temporal Consistent 4D Generation via Video Diffusion Models},
  author={Liang, Hanwen and Yin, Yuyang and Xu, Dejia and Liang, Hanxue and Wang, Zhangyang and Plataniotis, Konstantinos N and Zhao, Yao and Wei, Yunchao},
  journal={arXiv preprint arXiv:2405.16645},
  year={2024}
}

@article{zhang2024monst3r,
  title={MonST3R: A Simple Approach for Estimating Geometry in the Presence of Motion},
  author={Zhang, Junyi and Herrmann, Charles and Hur, Junhwa and Jampani, Varun and Darrell, Trevor and Cole, Forrester and Sun, Deqing and Yang, Ming-Hsuan},
  journal={arXiv preprint arxiv:2410.03825},
  year={2024}
}

@article{li20244k4dgen,
  title={4k4dgen: Panoramic 4d generation at 4k resolution},
  author={Li, Renjie and Pan, Panwang and Yang, Bangbang and Xu, Dejia and Zhou, Shijie and Zhang, Xuanyang and Li, Zeming and Kadambi, Achuta and Wang, Zhangyang and Tu, Zhengzhong and others},
  journal={arXiv preprint arXiv:2406.13527},
  year={2024}
}

@article{zhao2024genxd,
  author={Zhao, Yuyang and Lin, Chung-Ching and Lin, Kevin and Yan, Zhiwen and Li, Linjie and Yang, Zhengyuan and Wang, Jianfeng and Lee, Gim Hee and Wang, Lijuan},
  title={GenXD: Generating Any 3D and 4D Scenes},
  journal={arXiv preprint arXiv:2411.02319},
  year={2024}
}

@misc{cao2024avatargozeroshot4dhumanobject,
      title={AvatarGO: Zero-shot 4D Human-Object Interaction Generation and Animation}, 
      author={Yukang Cao and Liang Pan and Kai Han and Kwan-Yee K. Wong and Ziwei Liu},
      year={2024},
      eprint={2410.07164},
      archivePrefix={arXiv},
      primaryClass={cs.CV},
      url={https://arxiv.org/abs/2410.07164}, 
}

@article{sun2024eg4d,
  title={Eg4d: Explicit generation of 4d object without score distillation},
  author={Sun, Qi and Guo, Zhiyang and Wan, Ziyu and Yan, Jing Nathan and Yin, Shengming and Zhou, Wengang and Liao, Jing and Li, Houqiang},
  journal={arXiv preprint arXiv:2405.18132},
  year={2024}
}

@misc{fiebelman20244legs4dlanguageembedded,
โ€ƒ โ€ƒ title={4-LEGS: 4D Language Embedded Gaussian Splatting},
โ€ƒ โ€ƒ author={Gal Fiebelman and Tamir Cohen and Ayellet Morgenstern and Peter Hedman and Hadar Averbuch-Elor},
โ€ƒ โ€ƒ year={2024},
โ€ƒ โ€ƒ eprint={2410.10719},
โ€ƒ โ€ƒ archivePrefix={arXiv},
โ€ƒ โ€ƒ primaryClass={cs.CV}
}

@article{pang2024disco4d,
  title={Disco4D: Disentangled 4D Human Generation and Animation from a Single Image},
  author={Pang, Hui En and Liu, Shuai and Cai, Zhongang and Yang, Lei and Zhang, Tianwei and Liu, Ziwei},
  journal={arXiv preprint arXiv:2409.17280},
  year={2024}
}

@article{wu2024cat4d,
  title={Cat4d: Create anything in 4d with multi-view video diffusion models},
  author={Wu, Rundi and Gao, Ruiqi and Poole, Ben and Trevithick, Alex and Zheng, Changxi and Barron, Jonathan T and Holynski, Aleksander},
  journal={arXiv preprint arXiv:2411.18613},
  year={2024}
}

@article{wang20244real,
  title={4Real-Video: Learning Generalizable Photo-Realistic 4D Video Diffusion},
  author={Wang, Chaoyang and Zhuang, Peiye and Ngo, Tuan Duc and Menapace, Willi and Siarohin, Aliaksandr and Vasilkovsky, Michael and Skorokhodov, Ivan and Tulyakov, Sergey and Wonka, Peter and Lee, Hsin-Ying},
  journal={arXiv preprint arXiv:2412.04462},
  year={2024}
}

@article{chai2024star,
  title={Star: Skeleton-aware text-based 4d avatar generation with in-network motion retargeting},
  author={Chai, Zenghao and Tang, Chen and Wong, Yongkang and Kankanhalli, Mohan},
  journal={arXiv preprint arXiv:2406.04629},
  year={2024}
}

@inproceedings{som2024,
  title     = {Shape of Motion: 4D Reconstruction from a Single Video},
  author    = {Wang, Qianqian and Ye, Vickie and Gao, Hang and Austin, Jake and Li, Zhengqi and Kanazawa, Angjoo},
  journal   = {arXiv preprint arXiv:2407.13764},
  year      = {2024}
}

@article{sun2024dimensionx,
  title={Dimensionx: Create any 3d and 4d scenes from a single image with controllable video diffusion},
  author={Sun, Wenqiang and Chen, Shuo and Liu, Fangfu and Chen, Zilong and Duan, Yueqi and Zhang, Jun and Wang, Yikai},
  journal={arXiv preprint arXiv:2411.04928},
  year={2024}
}

@article{pan2026efficient4d,
  title={Efficient4d: Fast dynamic 3d object generation from a single-view video},
  author={Pan, Zijie and Yang, Zeyu and Zhu, Xiatian and Zhang, Li},
  journal={International Journal of Computer Vision},
  volume={134},
  number={1},
  pages={14},
  year={2026},
  publisher={Springer}
}

@InProceedings{Liang_2026_WACV,
    author    = {Liang, Hanwen and Xu, Dejia and Bhatt, Neel P. and Hu, Hezhen and Liang, Hanxue and Plataniotis, Konstantinos N.},
    title     = {Comp4D: Compositional 4D Scene Generation},
    booktitle = {Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)},
    month     = {March},
    year      = {2026},
    pages     = {3567-3577}
}

@inproceedings{li2026zerohsi,
  title={Zerohsi: Zero-shot 4d human-scene interaction by video generation},
  author={Li, Hongjie and Yu, Hong-Xing and Li, Jiaman and Wu, Jiajun},
  booktitle={2026 International Conference on 3D Vision (3DV)},
  pages={783--794},
  year={2026},
  organization={IEEE}
}


2024 ArXiv Papers

1. GaussianFlow: Splatting Gaussian Dynamics for 4D Content Creation

Quankai Gao, Qiangeng Xu, Zhe Cao, Ben Mildenhall, Wenchao Ma, Le Chen, Danhang Tang, Ulrich Neumann

(University of Southern California, Google, Pennsylvania State University, Max Planck Institute for Intelligent Systems)

Abstract Creating 4D fields of Gaussian Splatting from images or videos is a challenging task due to its under-constrained nature. While the optimization can draw photometric reference from the input videos or be regulated by generative models, directly supervising Gaussian motions remains underexplored. In this paper, we introduce a novel concept, Gaussian flow, which connects the dynamics of 3D Gaussians and pixel velocities between consecutive frames. The Gaussian flow can be efficiently obtained by splatting Gaussian dynamics into the image space. This differentiable process enables direct dynamic supervision from optical flow. Our method significantly benefits 4D dynamic content generation and 4D novel view synthesis with Gaussian Splatting, especially for contents with rich motions that are hard to be handled by existing methods. The common color drifting issue that happens in 4D generation is also resolved with improved Guassian dynamics. Superior visual quality on extensive experiments demonstrates our method's effectiveness. Quantitative and qualitative evaluations show that our method achieves state-of-the-art results on both tasks of 4D generation and 4D novel view synthesis.

2. PLA4D: Pixel-Level Alignments for Text-to-4D Gaussian Splatting

Qiaowei Miao, Yawei Luo, Yi Yang (Zhejiang University)

Abstract As text-conditioned diffusion models (DMs) achieve breakthroughs in image, video, and 3D generation, the research community's focus has shifted to the more challenging task of text-to-4D synthesis, which introduces a temporal dimension to generate dynamic 3D objects. In this context, we identify Score Distillation Sampling (SDS), a widely used technique for text-to-3D synthesis, as a significant hindrance to text-to-4D performance due to its Janus-faced and texture-unrealistic problems coupled with high computational costs. In this paper, we propose Pixel-Level Alignments for Text-to-4D Gaussian Splatting (PLA4D), a novel method that utilizes text-to-video frames as explicit pixel alignment targets to generate static 3D objects and inject motion into them. Specifically, we introduce Focal Alignment to calibrate camera poses for rendering and GS-Mesh Contrastive Learning to distill geometry priors from rendered image contrasts at the pixel level. Additionally, we develop Motion Alignment using a deformation network to drive changes in Gaussians and implement Reference Refinement for smooth 4D object surfaces. These techniques enable 4D Gaussian Splatting to align geometry, texture, and motion with generated videos at the pixel level. Compared to previous methods, PLA4D produces synthesized outputs with better texture details in less time and effectively mitigates the Janus-faced problem. PLA4D is fully implemented using open-source models, offering an accessible, user-friendly, and promising direction for 4D digital content creation.

3. 4Dynamic: Text-to-4D Generation with Hybrid Priors

Yu-Jie Yuan, Leif Kobbelt, Jiwen Liu, Yuan Zhang, Pengfei Wan, Yu-Kun Lai, Lin Gao

Abstract Due to the fascinating generative performance of text-to-image diffusion models, growing text-to-3D generation works explore distilling the 2D generative priors into 3D, using the score distillation sampling (SDS) loss, to bypass the data scarcity problem. The existing text-to-3D methods have achieved promising results in realism and 3D consistency, but text-to-4D generation still faces challenges, including lack of realism and insufficient dynamic motions. In this paper, we propose a novel method for text-to-4D generation, which ensures the dynamic amplitude and authenticity through direct supervision provided by a video prior. Specifically, we adopt a text-to-video diffusion model to generate a reference video and divide 4D generation into two stages: static generation and dynamic generation. The static 3D generation is achieved under the guidance of the input text and the first frame of the reference video, while in the dynamic generation stage, we introduce a customized SDS loss to ensure multi-view consistency, a video-based SDS loss to improve temporal consistency, and most importantly, direct priors from the reference video to ensure the quality of geometry and texture. Moreover, we design a prior-switching training strategy to avoid conflicts between different priors and fully leverage the benefits of each prior. In addition, to enrich the generated motion, we further introduce a dynamic modeling representation composed of a deformation network and a topology network, which ensures dynamic continuity while modeling topological changes. Our method not only supports text-to-4D generation but also enables 4D generation from monocular videos. The comparison experiments demonstrate the superiority of our method compared to existing methods.

4. CT4D: Consistent Text-to-4D Generation with Animatable Meshes

Ce Chen, Shaoli Huang, Xuelin Chen, Guangyi Chen, Xiaoguang Han, Kun Zhang, Mingming Gong

(Mohamed bin Zayed University of Artificial Intelligence, Tencent AI Lab, Carnegie Mellon University, FNii CUHKSZ, SSE CUHKSZ, University of Melbourne)

Abstract Text-to-4D generation has recently been demonstrated viable by integrating a 2D image diffusion model with a video diffusion model. However, existing models tend to produce results with inconsistent motions and geometric structures over time. To this end, we present a novel framework, coined CT4D, which directly operates on animatable meshes for generating consistent 4D content from arbitrary user-supplied prompts. The primary challenges of our mesh-based framework involve stably generating a mesh with details that align with the text prompt while directly driving it and maintaining surface continuity. Our CT4D framework incorporates a unique Generate-Refine-Animate (GRA) algorithm to enhance the creation of text-aligned meshes. To improve surface continuity, we divide a mesh into several smaller regions and implement a uniform driving function within each area. Additionally, we constrain the animating stage with a rigidity regulation to ensure cross-region continuity. Our experimental results, both qualitative and quantitative, demonstrate that our CT4D framework surpasses existing text-to-4D techniques in maintaining interframe consistency and preserving global geometry. Furthermore, we showcase that this enhanced representation inherently possesses the capability for combinational 4D generation and texture editing.

5. Trans4D: Realistic Geometry-Aware Transition for Compositional Text-to-4D Synthesis

Bohan Zeng, Ling Yang, Siyu Li, Jiaming Liu, Zixiang Zhang, Juanxi Tian, Kaixin Zhu, Yongzhen Guo, Fu-Yun Wang, Minkai Xu, Stefano Ermon, Wentao Zhang

(Peking University, The Chinese University of Hong Kong, Stanford University)

Abstract Recent advances in diffusion models have demonstrated exceptional capabilities in image and video generation, further improving the effectiveness of 4D synthesis. Existing 4D generation methods can generate high-quality 4D objects or scenes based on user-friendly conditions, benefiting the gaming and video industries. However, these methods struggle to synthesize significant object deformation of complex 4D transitions and interactions within scenes. To address this challenge, we propose Trans4D, a novel text-to-4D synthesis framework that enables realistic complex scene transitions. Specifically, we first use multi-modal large language models (MLLMs) to produce a physic-aware scene description for 4D scene initialization and effective transition timing planning. Then we propose a geometry-aware 4D transition network to realize a complex scene-level 4D transition based on the plan, which involves expressive geometrical object deformation. Extensive experiments demonstrate that Trans4D consistently outperforms existing state-of-the-art methods in generating 4D scenes with accurate and high-quality transitions, validating its effectiveness.

6. PaintScene4D: Consistent 4D Scene Generation from Text Prompts

Vinayak Gupta, Yunze Man, Yu-Xiong Wang

(Indian Institute of Technology Madras, University of Illinois Urbana-Champaign)

Abstract Recent advances in diffusion models have revolutionized 2D and 3D content creation, yet generating photorealistic dynamic 4D scenes remains a significant challenge. Existing dynamic 4D generation methods typically rely on distilling knowledge from pre-trained 3D generative models, often fine-tuned on synthetic object datasets. Consequently, the resulting scenes tend to be object-centric and lack photorealism. While text-to-video models can generate more realistic scenes with motion, they often struggle with spatial understanding and provide limited control over camera viewpoints during rendering. To address these limitations, we present PaintScene4D, a novel text-to-4D scene generation framework that departs from conventional multi-view generative models in favor of a streamlined architecture that harnesses video generative models trained on diverse real-world datasets. Our method first generates a reference video using a video generation model, and then employs a strategic camera array selection for rendering. We apply a progressive warping and inpainting technique to ensure both spatial and temporal consistency across multiple viewpoints. Finally, we optimize multi-view images using a dynamic renderer, enabling flexible camera control based on user preferences. Adopting a training-free architecture, our PaintScene4D efficiently produces realistic 4D scenes that can be viewed from arbitrary trajectories.

7. Bringing Objects to Life: 4D generation from 3D objects

Ohad Rahamim, Ori Malca, Dvir Samuel, Gal Chechik (Bar-Ilan University, NVIDIA)

Abstract Recent advancements in generative modeling now enable the creation of 4D content (moving 3D objects) controlled with text prompts. 4D generation has large potential in applications like virtual worlds, media, and gaming, but existing methods provide limited control over the appearance and geometry of generated content. In this work, we introduce a method for animating user-provided 3D objects by conditioning on textual prompts to guide 4D generation, enabling custom animations while maintaining the identity of the original object. We first convert a 3D mesh into a ``static" 4D Neural Radiance Field (NeRF) that preserves the visual attributes of the input object. Then, we animate the object using an Image-to-Video diffusion model driven by text. To improve motion realism, we introduce an incremental viewpoint selection protocol for sampling perspectives to promote lifelike movement and a masked Score Distillation Sampling (SDS) loss, which leverages attention maps to focus optimization on relevant regions. We evaluate our model in terms of temporal coherence, prompt adherence, and visual fidelity and find that our method outperforms baselines that are based on other approaches, achieving up to threefold improvements in identity preservation measured using LPIPS scores, and effectively balancing visual quality with dynamic content.
YearTitleArXiv TimePaperCodeProject Page
2024GaussianFlow: Splatting Gaussian Dynamics for 4D Content Creation19 Mar 2024LinkLinkLink
2024PLA4D: Pixel-Level Alignments for Text-to-4D Gaussian Splatting4 Jun 2024Link--Link
20244Dynamic: Text-to-4D Generation with Hybrid Priors17 Jul 2024Link----
2024CT4D: Consistent Text-to-4D Generation with Animatable Meshes15 Aug 2024Link----
2024Trans4D: Realistic Geometry-Aware Transition for Compositional Text-to-4D Synthesis9 Oct 2024LinkLink--
2024PaintScene4D: Consistent 4D Scene Generation from Text Prompts5 Dec 2024LinkLinkLink
2024Bringing Objects to Life: 4D generation from 3D objects29 Dec 2024LinkLinkLink
ArXiv Papers References
%axiv papers

@article{gao2024gaussianflow,
  title={GaussianFlow: Splatting Gaussian Dynamics for 4D Content Creation},
  author={Gao, Quankai and Xu, Qiangeng and Cao, Zhe and Mildenhall, Ben and Ma, Wenchao and Chen, Le and Tang, Danhang and Neumann, Ulrich},
  journal={arXiv preprint arXiv:2403.12365},
  year={2024}
}

@misc{miao2024pla4d,
      title={PLA4D: Pixel-Level Alignments for Text-to-4D Gaussian Splatting}, 
      author={Qiaowei Miao and Yawei Luo and Yi Yang},
      year={2024},
      eprint={2405.19957},
      archivePrefix={arXiv},
      primaryClass={cs.CV}
}

@misc{yuan20244dynamictextto4dgenerationhybrid,
      title={4Dynamic: Text-to-4D Generation with Hybrid Priors}, 
      author={Yu-Jie Yuan and Leif Kobbelt and Jiwen Liu and Yuan Zhang and Pengfei Wan and Yu-Kun Lai and Lin Gao},
      year={2024},
      eprint={2407.12684},
      archivePrefix={arXiv},
      primaryClass={cs.CV},
      url={https://arxiv.org/abs/2407.12684}, 
}

@misc{chen2024ct4dconsistenttextto4dgeneration,
      title={CT4D: Consistent Text-to-4D Generation with Animatable Meshes}, 
      author={Ce Chen and Shaoli Huang and Xuelin Chen and Guangyi Chen and Xiaoguang Han and Kun Zhang and Mingming Gong},
      year={2024},
      eprint={2408.08342},
      archivePrefix={arXiv},
      primaryClass={cs.GR},
      url={https://arxiv.org/abs/2408.08342}, 
}

@article{zeng2024trans4d,
  title={Trans4D: Realistic Geometry-Aware Transition for Compositional Text-to-4D Synthesis},
  author={Zeng, Bohan and Yang, Ling and Li, Siyu and Liu, Jiaming and Zhang, Zixiang and  Tian, Juanxi and Zhu, Kaixin and Guo, Yongzhen and Wang, Fu-Yun and Xu, Minkai and Ermon, Stefano and Zhang, Wentao},
  journal={arXiv preprint arXiv:2410.07155},
  year={2024}
}

@article{gupta2024paintscene4d,
title={PaintScene4D: Consistent 4D Scene Generation from Text Prompts},
author={Gupta, Vinayak and Man, Yunze and Wang, Yuxiong},
journal={https://arxiv.org/abs/2412.04471},
year={2024}
}

@article{rahamim2024bringingobjectslife4d,
      title={Bringing Objects to Life: 4D generation from 3D objects}, 
      author={Ohad Rahamim and Ori Malca and Dvir Samuel and Gal Chechik},
      year={2024},
      eprint={2412.20422},
      archivePrefix={arXiv},
      primaryClass={cs.CV},
      url={https://arxiv.org/abs/2412.20422}, 
}