[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-128445-en":3,"doc-seo-128445-105":30,"detail-sidebar-cat-0-en-105":92},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":20,"is_deleted":4,"is_public":20,"is_downloadable":20,"audit_status":20,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":14,"update_tm":28,"read_time":29},128445,962085662650,"Jiven","https://ap-avatar.wpscdn.com/davatar_29158cc5080c5b710cf443261637dec0",8,"Research & Report","Adaptation of Speaker and Speech Recognition Methods for the Automatic Screening of Speech Disorders using Machine Learning","This Ph.D. thesis investigates machine-learning approaches that adapt speaker and automatic speech recognition methods to support automatic screening of speech disorders. It reviews speech and speaker recognition, paralinguistics, and pathological speech processing, then develops feature-based pipelines including i-vector and Fisher Vector representations, as well as deep neural embeddings such as x-vector. The work covers dataset-specific experimentation (e.g., corpora for neurological, respiratory, sleepiness, and depression screening) and evaluates utterance-level modeling using temporal speech parameters and posterior-thresholding hesitation representations, followed by result discussion and concluding remarks.","1  \n1.5em  \nAdaptation of Speaker and Speech Recognition Methods for the Automatic Screening of Speech Disorders using Machine Learning  \nPh.D. Thesis  \nJos Vicente Egas Lpez Supervisor: Gbor Gosztolya, Ph.D.  \nDoctoral School of Computer Science Department of Computer Algorithms and Artiﬁcial Intelligence Faculty of Science and Informatics University of Szeged  \nSzeged, October 16, 2022  \n4  \nContents  \n1 Introduction 5  \n1.1 Speech and Speaker Recognition: A Brief Review ............ 6  \n1.1.1 Automatic Speech Recognition .................. 6  \n1.1.2 Speaker Recognition ........................ 8  \n1.2 Paralanguage ................................ 9  \n1.2.1 Computational Paralinguistics ................... 9  \n1.2.2 Contemporary Research ...................... 10  \n1.3 Structure of the Dissertation ........................ 10  \n2 Machine Learning and Pathological Speech Processing 13  \n2.1 Machine Learning .............................. 13  \n2.1.1 Supervised Machine Learning ................... 14  \n2.1.2 Support Vector Machines ...................... 16  \n2.1.3 XGBoost Algorithm ......................... 16  \n2.2 Pathological Speech Processing ...................... 17  \n2.2.1 Corpora and Feature Representations ............... 18  \n2.2.2 Frame-level Features ........................ 19  \n3 Front-End Factor Analysis 23  \n3.1 Introduction ................................. 23  \n3.2 Related Works ................................ 24  \n3.3 The i-vector Approach ........................... 25  \n3.4 The Corpora ................................. 25  \n3.5 The Experiments .............................. 26  \n3.5.1 Feature Extraction ......................... 26  \n3.5.2 The i-vector Training ........................ 26  \n3.5.3 Evaluation .............................. 27  \n3.5.4 Results ................................ 28  \n3.6 Concluding remarks ............................ 29  \n4 The Fisher Vector 31  \n4.1 Introduction ................................. 31  \n4.1.1 Parkinson's Disease Screening ................... 32  \n4.1.2 Cold Speech Screening ....................... 32  \n4.1.3 Escalation in the Dialogue, and Primates Species Detection ... 33  \n4.2 Related Works ................................ 33  \n4.3 The Fisher Vector .............................. 34  \n4.3.1 The Fisher Kernel .......................... 34  \n4.3.2 The Fisher Vector for audio-signals ................ 35  \n4.4 The Corpora ................................. 36  \n4.4.1 PC-GITA Corpus .......................... 36  \n4.4.2 Upper Respiratory Tract Infection Corpus (URTIC) ....... 36  \n4.4.3 Escalation Corpus ......................... 37  \n4.4.4 Primates Vocalisation Corpus ................... 37  \n4.5 The Experiments .............................. 38  \n4.5.1 Feature Extraction ......................... 38  \n4.5.2 Training and Evaluation Methods ................. 39  \n4.5.3 Results and Discussion ....................... 41  \n4.6 Concluding remarks ............................ 46  \n5 Deep Neural Network Embeddings 49  \n5.1 The x-vector Method ............................ 49  \n5.1.1 DNN structure ........................... 49  \n5.1.2 Embeddings ............................. 50  \n5.2 Excessive Daytime Sleepiness Detection ................. 50  \n5.2.1 SLEEP (Dusseldorf Sleepy Language) Corpus .......... 51  \n5.2.2 Related Works ........................... 52  \n5.3 Clinical Depression Screening ....................... 52  \n5.3.1 Hungarian Depressed Speech Dataset (HDSDb) ......... 52  \n5.3.2 Related Works ........................... 53  \n5.4 Escalation and Primates .......................... 53  \n5.4.1 Escalation and Primates Corpora ................. 53  \n5.5 The Experiments .............................. 53  \n5.5.1 Feature Extraction ......................... 54  \n5.5.2 Training and Evaluation Methods ................. 54  \n5.6 Results and Discussion ........................... 56  \n5.6.1 Sleepiness .............................. 56  \n5.6.2 Depression ......................","cbCairjqvhBLw5Cj","https://ap.wps.com/l/cbCairjqvhBLw5Cj","pdf",3394128,1,131,"English","en",105,"# Contents\n## Introduction\n## Machine Learning and Pathological Speech Processing\n## Front-End Factor Analysis\n## The Fisher Vector\n## Deep Neural Network Embeddings\n## Automatic Speech Recognition Methods\n## Bibliography\n## Summary\n## Publications","[{\"question\":\"What is the main goal of the thesis?\",\"answer\":\"The thesis aims to adapt speaker and speech recognition methods, using machine learning, for automatic screening of speech disorders.\"},{\"question\":\"Which core feature representations are investigated?\",\"answer\":\"It studies i-vector front-end factor analysis and Fisher Vector representations, and it also explores deep neural embeddings such as x-vector.\"},{\"question\":\"How are speech disorder screening tasks evaluated?\",\"answer\":\"Evaluation is performed using experiments built around corpus-specific training and testing, with utterance-level feature extraction based on temporal speech parameters and posterior-thresholding hesitation representations.\"}]","Adaptation of Speaker and Speech Recognition Methods for the Automatic Screening of Speech Disorders using Machine Learning | PDF",1786001101,330,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":13,"keywords":34,"description":14,"schema_data":35,"social_meta":87,"head_meta":89,"extra_data":91,"updated_unix":28},"adaptation-of-speaker-and-speech-recognition-methods-for-the-automatic-screening-of-speech-disorders-using-machine-learning","",{"@graph":36,"@context":86},[37,54,69],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":20},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/document/","Document",2,{"item":49,"name":12,"@type":43,"position":50},"https://docshare.wps.com/document/research-report/",3,{"item":52,"name":13,"@type":43,"position":53},"https://docshare.wps.com/document/adaptation-of-speaker-and-speech-recognition-methods-for-the-automatic-screening-of-speech-disorders-using-machine-learning/128445/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":23,"description":14,"dateModified":62,"datePublished":63,"encodingFormat":61,"isAccessibleForFree":64,"interactionStatistic":65},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-08-22","2026-08-06",true,{"@type":66,"interactionType":67,"userInteractionCount":20},"InteractionCounter",{"@type":68},"ViewAction",{"@type":70,"mainEntity":71},"FAQPage",[72,78,82],{"name":73,"@type":74,"acceptedAnswer":75},"What is the main goal of the thesis?","Question",{"text":76,"@type":77},"The thesis aims to adapt speaker and speech recognition methods, using machine learning, for automatic screening of speech disorders.","Answer",{"name":79,"@type":74,"acceptedAnswer":80},"Which core feature representations are investigated?",{"text":81,"@type":77},"It studies i-vector front-end factor analysis and Fisher Vector representations, and it also explores deep neural embeddings such as x-vector.",{"name":83,"@type":74,"acceptedAnswer":84},"How are speech disorder screening tasks evaluated?",{"text":85,"@type":77},"Evaluation is performed using experiments built around corpus-specific training and testing, with utterance-level feature extraction based on temporal speech parameters and posterior-thresholding hesitation representations.","https://schema.org",{"og:url":52,"og:type":88,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":90,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":93},[94,98,102,106,111,116,121,124,129,132,136],{"id":20,"doc_module":4,"doc_module_name":46,"category_name":95,"show_sort_weight":96,"slug":97},"Story & Novel",90,"story-novel",{"id":47,"doc_module":4,"doc_module_name":46,"category_name":99,"show_sort_weight":100,"slug":101},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":46,"category_name":103,"show_sort_weight":104,"slug":105},"Exam",70,"exam",{"id":107,"doc_module":4,"doc_module_name":46,"category_name":108,"show_sort_weight":109,"slug":110},5,"Comic",60,"comic",{"id":112,"doc_module":4,"doc_module_name":46,"category_name":113,"show_sort_weight":114,"slug":115},6,"Technology",50,"technology",{"id":117,"doc_module":4,"doc_module_name":46,"category_name":118,"show_sort_weight":119,"slug":120},7,"Healthcare",40,"healthcare",{"id":11,"doc_module":4,"doc_module_name":46,"category_name":12,"show_sort_weight":122,"slug":123},30,"research-report",{"id":125,"doc_module":4,"doc_module_name":46,"category_name":126,"show_sort_weight":127,"slug":128},9,"Religion & Spirituality",20,"religion-spirituality",{"id":127,"doc_module":4,"doc_module_name":46,"category_name":130,"show_sort_weight":127,"slug":131},"World Cup","world-cup",{"id":133,"doc_module":4,"doc_module_name":46,"category_name":134,"show_sort_weight":133,"slug":135},10,"Lifestyle","lifestyle",{"id":137,"doc_module":4,"doc_module_name":46,"category_name":138,"show_sort_weight":107,"slug":139},19,"General","general"]