[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-137585-105":59,"doc-detail-137585-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","atrie-adaptive-tuning-for-robust-inference-and-emotion-in-persona-driven-speech-synthesis","ATRIE - Adaptive Tuning for Robust Inference and Emotion in Persona-Driven Speech Synthesis","","High-fidelity character voice synthesis underpins immersive multimedia, especially when interacting with anime avatars and digital humans. Existing approaches cannot keep persona traits consistent across varied emotional conditions. ATRIE introduces a Persona-Prosody DualTrack (P2-DT) framework that separates a static timbre representation from a dynamic prosody representation, distilled from a 14B LLM teacher, enabling stable identity preservation, robust zero-shot verification, and expressive emotional output.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/atrie-adaptive-tuning-for-robust-inference-and-emotion-in-persona-driven-speech-synthesis/137585/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/atrie-adaptive-tuning-for-robust-inference-and-emotion-in-persona-driven-speech-synthesis/137585.png","ImageObject",300,407,{"name":92,"@type":93},"Levi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-20","2026-08-22",true,{"@type":102,"interactionType":103,"userInteractionCount":24},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does ATRIE address in character voice synthesis?","Question",{"text":112,"@type":113},"ATRIE targets inconsistent persona traits across different emotional contexts, where prior systems tend to be either emotionally flat or character-inconsistent.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does ATRIE model persona and emotion?",{"text":117,"@type":113},"It uses a Persona-Prosody DualTrack (P2-DT) architecture with a static timbre track and a dynamic prosody track, enabling persona-consistent emotion expression.",{"name":119,"@type":110,"acceptedAnswer":120},"What performance indicators are reported for ATRIE?",{"text":121,"@type":113},"ATRIE reports strong results on its extended AnimeTTS-Bench, including identity preservation with low EER and state-of-the-art generation and crossmodal retrieval performance.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},137585,1787422181,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":24,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":52,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":129,"read_time":143},7971461740909,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","ATRIE: Adaptive Tuning for Robust Inference and Emotion in  \nPersona-Driven Speech Synthesis  \nAoduo Li  \nGuangdong University of Technology Guangzhou, China [3123009124@mail2.gdut.edu.cn](3123009124@mail2.gdut.edu.cn)  \nHaoran Lv  \nGuangdong University of Technology Guangzhou, China [3123008610@mail2.gdut.edu.cn](3123008610@mail2.gdut.edu.cn)  \nHongjian Xu  \nGuangdong University of Technology Guangzhou, China [123457890wasd@gmail.com](123457890wasd@gmail.com)  \narXiv :2604 . 19055v2 [ cs . SD] 23 Apr 2026  \nShengmin Li  \nSouth China University of Technology Guangzhou, China [milishengmin_@mail.scut.edu.cn](milishengmin_@mail.scut.edu.cn)  \nSihao Qin  \nSouth China University of Technology Guangzhou, China [202330363461@mail.scut.edu.cn](202330363461@mail.scut.edu.cn)  \nZimeng Li  \nShenzhen Polytechnic University Shenzhen, China [li_zimeng@szpu.edu.cn](li_zimeng@szpu.edu.cn)  \nChi Man Pun  \nUniversity of Macau Macau, China [cmpun@umac.mo](cmpun@umac.mo)  \nXuhang Chen∗ Huizhou University Huizhou, China [xuhangc@hzu.edu.cn](xuhangc@hzu.edu.cn)  \nAbstract  \nHigh-fidelity character voice synthesis is a cornerstone of immersive multimedia applications, particularly for interacting with anime avatars and digital humans. However, existing systems struggle to maintain consistent persona traits across diverse emotional contexts. To bridge this gap, we present ATRIE, a unified framework utilizing a Persona-Prosody DualTrack (P2-DT) architecture. Our system disentangles generation into a static Timbre Track (via Scalar Quantization) anda dynamic Prosody Track (via Hierarchical Flow-Matching), distilled from a 14B LLM teacher. This design enables robust identity preservation (Zero-Shot Speaker Verification EER: 0.04) and rich emotional expression. Evaluated on our extended AnimeTTS-Bench (50 characters), ATRIE achieves state-of-the-art performance in both generation and crossmodal retrieval (mAP: 0.75), establishing a new paradigm for persona-driven multimedia content creation.  \nCCS Concepts  \n• Computing methodologies → Natural language generation;  \n• Applied computing → Sound and music computing.  \n∗ Corresponding author.  \nPermission to make digital or hard copies of all or part of this work for personal or classroom use is granted without fee provided that copies are not made or distributed for profit or commercial advantage and that copies bear this notice and the full citation on the first page. Copyrights for components of this work owned by others than the author(s) must be honored. Abstracting with credit is permitted. To copy otherwise, or republish, to post on servers or to redistribute to lists, requires prior specific permission and/or a fee. Request permissions [from permissions@acm.org](from permissions@acm.org).  \nConference’17, Washington, DC, USA  \n© 2026 Copyright held by the owner/author(s) . Publication rights licensed to ACM.  \nACM ISBN 978-x-xxxx-xxxx-x/YYYY/MM [https://doi.org/10.1145/nnnnnnn.nnnnnnn](https://doi.org/10.1145/nnnnnnn.nnnnnnn)  \nKeywords  \nText-to-Speech, Anime Characters, Large Language Models, Persona Understanding, Emotional Expression  \nACM Reference Format:  \nAoduo Li, Haoran Lv, Hongjian Xu, Shengmin Li, Sihao Qin, Zimeng Li, Chi Man Pun, and Xuhang Chen. 2026. ATRIE: Adaptive Tuning for Robust Inference and Emotion in Persona-Driven Speech Synthesis. In . ACM, New York, NY, USA, 10 pages. [https://doi.org/10.1145/nnnnnnn.nnnnnnn](https://doi.org/10.1145/nnnnnnn.nnnnnnn)  \n1 Introduction  \nThe rapid expansion of virtual characters in consumer electronics—spanning video game companions, virtual streaming avatars (VTubers), and intelligent assistants—has generated unprecedented demand for personalized, expressive voice synthesis systems. In 2023 alone, the global VTuber market was valued at over $2.5 billion, underscoring the critical need for high-fidelity character voice generation. Among these applications, anime character voice synthesis presents particularly stringent requirements: users ex","cbCaiaD1IGpRvXxj","https://ap.wps.com/l/cbCaiaD1IGpRvXxj","pdf",1817123,"English","# Abstract\n# 1 Introduction\n## 1.1 Motivation and Challenges\n## 1.2 Pro","[{\"question\":\"What problem does ATRIE address in character voice synthesis?\",\"answer\":\"ATRIE targets inconsistent persona traits across different emotional contexts, where prior systems tend to be either emotionally flat or character-inconsistent.\"},{\"question\":\"How does ATRIE model persona and emotion?\",\"answer\":\"It uses a Persona-Prosody DualTrack (P2-DT) architecture with a static timbre track and a dynamic prosody track, enabling persona-consistent emotion expression.\"},{\"question\":\"What performance indicators are reported for ATRIE?\",\"answer\":\"ATRIE reports strong results on its extended AnimeTTS-Bench, including identity preservation with low EER and state-of-the-art generation and crossmodal retrieval performance.\"}]","ATRIE - Adaptive Tuning for Robust Inference and Emotion in Persona-Driven Speech Synthesis | PDF",25]