[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-141731-105":59,"doc-detail-141731-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","dreamwaltz-make-a-scene-with-complex-3d-animatable-avatars","DreamWaltz - Make a Scene with Complex 3DAnimatable Avatars","","DreamWaltz提出一种从文本引导生成并驱动复杂三维可动画头像的新框架。针对现有文本到三维方法难以同时获得高质量、可动画结构一致性的问题，方法引入3D一致、具遮挡感知的Score Distillation Sampling，并以规范姿态优化隐式神经表示。通过3D感知骨架条件提供视角对齐监督，减少伪影与多面问题。动画阶段利用扩散模型在多姿态图像先验上学习可动画头像表征，实现无需重训练的任意姿态变形与场景合成。",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/dreamwaltz-make-a-scene-with-complex-3d-animatable-avatars/141731/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/dreamwaltz-make-a-scene-with-complex-3d-animatable-avatars/141731.png","ImageObject",300,407,{"name":92,"@type":93},"Eliana","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-19","2026-08-25",true,{"@type":102,"interactionType":103,"userInteractionCount":24},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"DreamWaltz解决了文本生成3D头像中的哪些核心挑战？","Question",{"text":112,"@type":113},"它针对复杂外观细节、可关节姿态一致性以及不同姿态下形状与纹理变化带来的动画难题，提出能生成可动画头像的框架。","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"DreamWaltz在头像生成阶段使用了什么关键方法？",{"text":117,"@type":113},"生成阶段采用3D-consistent且aware occlusion的Score Distillation Sampling（SDS），并结合规范姿态与3D感知骨架条件进行视角对齐监督。",{"name":119,"@type":110,"acceptedAnswer":120},"DreamWaltz如何实现无需重训练的复杂非rigged头像动画？",{"text":121,"@type":113},"通过在姿态先验条件下训练可动画NeRF表征，使生成的头像能在测试时变形到任意姿态，从而实现真实动画而不需要重新训练。","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},141731,1787663683,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":24,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":56,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":129,"read_time":143},4398048949847,"https://ap-avatar.wpscdn.com/avatar/400002536579ef2da7f?_k=1778318612642679267","DreamWaltz: Make a Scene with Complex 3DAnimatable Avatars  \nYukun Huang 1 ,2∗†, Jianan Wang 1∗‡, Ailing Zeng 1 , He Cao 1 , Xianbiao Qi 1 , Yukai Shi 1 ,  \nZheng-Jun Zha2 , Lei Zhang 1  \n1International Digital Economy Academy (IDEA)  \n2University of Science and Technology of China  \n(a) Princess Elsa in different poses. (b) Uzumaki Naruto in dance moves.  \n(c) Woody dances with Stormtrooper. (d) Woody sits on: (Left) a chair; (Right) a chair made of cheese.  \nFigure 1: DreamWaltz is a text-to-3D-avatar generation framework, which can (a, b) create complex 3D animatable avatars from texts,(c, d) ready for 3D scene composition with diverse interactions.  \nAbstract  \nWe present DreamWaltz, a novel framework for generating and animating complex 3D avatars given text guidance and parametric human body prior. While recent methods have shown encouraging results for text-to-3D generation of common objects, creating high-quality and animatable 3D avatars remains challenging. To create high-quality 3D avatars, DreamWaltz proposes 3D-consistent occlusionaware Score Distillation Sampling (SDS) to optimize implicit neural representations with canonical poses. It provides view-aligned supervision via 3D-aware skeleton conditioning which enables complex avatar generation without artifacts and multiple faces. For animation, our method learns an animatable 3D avatar representation from abundant image priors of diffusion model conditioned on various poses, which could animate complex non-rigged avatars given arbitrary poses without retraining.  \nExtensive evaluations demonstrate that DreamWaltz is an effective and robust approach for creating 3D avatars that can take on complex shapes and appearances as well as novel poses for animation. The proposed framework further enables the creation of complex scenes with diverse compositions, including avatar-avatar, avatar-object and avatar-scene interactions. See [https://dreamwaltz3d.github.io/](https://dreamwaltz3d.github.io/ for)[ for](https://dreamwaltz3d.github.io/ for)[ ](https://dreamwaltz3d.github.io/ for)more vivid 3D avatar and animation results.  \n1 Introduction  \nThe creation and animation of 3D digital avatars are essential for various applications, including film and cartoon production, video game design, and immersive media such as AR and VR. However,  \n∗Equal contribution.  \n†Work done during an internship at IDEA.  \n‡Corresponding author.  \n37th Conference on Neural Information Processing Systems (NeurIPS 2023) .  \ntraditional techniques for constructing such intricate 3D models are costly and time-consuming, requiring thousands of hours from skilled artists with extensive aesthetics and 3D modeling knowledge. In this work, we seek a solution for 3D avatar generation that satisfies the following desiderata: (1) easily controllable over avatar properties through textual descriptions; (2) capable of producing high-quality and diverse 3D avatars with complex shapes and appearances; (3) the generated avatars should be ready for animation and scene composition with diverse interactions.  \nThe advancement of deep learning methods has enabled promising methods which can reconstruct 3D human models from monocular images [36, 45] or videos [44, 14, 46, 41, 12, 30] . Nonetheless, these methods rely heavily on the strong visual priors from image/video and human body geometry, making them unsuitable for generating creative avatars that can take on complex and imaginative shapes or appearances. Recently, integrating 2D generative models into 3D modeling [31, 18, 10] has gained significant attention to make 3D digitization more accessible, reducing the dependency on extensive 3D datasets. However, creating animatable 3D avatars remains challenging: Firstly, avatars often require intricate and complex details for their appearance (e.g., loose cloth, diverse hair, and different accessories); secondly, avatars have articulated structures where each body part is able to assume various poses in a coordi","cbCaieTAjXIfuxtv","https://ap.wps.com/l/cbCaieTAjXIfuxtv","pdf",13066114,"English","# Introduction\n## Problem and motivation\n## Proposed framework overview\n## Contributions","[{\"question\":\"DreamWaltz解决了文本生成3D头像中的哪些核心挑战？\",\"answer\":\"它针对复杂外观细节、可关节姿态一致性以及不同姿态下形状与纹理变化带来的动画难题，提出能生成可动画头像的框架。\"},{\"question\":\"DreamWaltz在头像生成阶段使用了什么关键方法？\",\"answer\":\"生成阶段采用3D-consistent且aware occlusion的Score Distillation Sampling（SDS），并结合规范姿态与3D感知骨架条件进行视角对齐监督。\"},{\"question\":\"DreamWaltz如何实现无需重训练的复杂非rigged头像动画？\",\"answer\":\"通过在姿态先验条件下训练可动画NeRF表征，使生成的头像能在测试时变形到任意姿态，从而实现真实动画而不需要重新训练。\"}]","DreamWaltz - Make a Scene with Complex 3DAnimatable Avatars | PDF",48]