[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-125373-en":3,"doc-seo-125373-105":30,"detail-sidebar-cat-0-en-105":91},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":4,"is_deleted":4,"is_public":20,"is_downloadable":20,"audit_status":20,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":14,"update_tm":28,"read_time":29},125373,13056703019662,"Evangeline","https://ap-avatar.wpscdn.com/avatar/be000253a8e92610077?_k=1778726343310543188",8,"Research & Report","Context-Aware Real-Time Audio Machine Learning - Dissertation","Context-aware real-time audio machine learning enables robust sound understanding under changing environmental conditions and streaming constraints. This dissertation presents methods for sound event detection in field recordings, emphasizing data collection, spectrogram-based representations, convolutional classification, and post-classification corrections, including comparisons to BirdNET. It further quantifies soundscape components related to anthropophony, geophony, and biophony using deep learning and statistical analyses, and introduces target sound extraction approaches spanning multi-task extraction, reverberant settings, and real-time causal processing, with experiments and ablation studies validating effectiveness.","UC Merced  \nUC Merced Electronic Theses and Dissertations  \nTitle  \nContext-Aware Real-Time Audio Machine Learning  \nPermalink  \n[https://escholarship.org/uc/item/0h52z58g](https://escholarship.org/uc/item/0h52z58g)  \nISBN  \n9798297603691  \nAuthor  \nBaligar, Shrishail Kishor  \nPublication Date  \n2025-08-11  \nPeer reviewed|Thesis/dissertation  \n[eScholarship.org](eScholarship.org) Powered by the California Digital Library  \nUniversity of California  \nUNIVERSITY OF CALIFORNIA, MERCED  \nContext-Aware Real-Time Audio Machine Learning  \nA dissertation submitted in partial satisfaction of the requirements for the degree  \nDoctor of Philosophy  \nin  \nElectrical Engineering and Computer Science  \nby  \nShrishail Kishor Baligar  \nCommittee in charge:  \nProfessor Shawn Newsam, Chair  \nProfessor Arif Ahmed  \nProfessor Shijia Pan  \n2025  \nCopyright  \nShrishail Kishor Baligar, 2025 All rights reserved.  \nThe dissertation of Shrishail Kishor Baligar is approved, and it is acceptable in quality and form for publication on microfilm and electronically:  \n(Professor Arif Ahmed)  \n(Professor Shijia Pan)  \n(Professor Shawn Newsam, Chair)  \nUniversity of California, Merced  \n2025  \nDEDICATION  \nTo UC Merced  \nTABLE OF CONTENTS  \nSignature Page .......................... iii  \nDedication ............................. iv  \nTable of Contents ......................... v  \nList of Figures ........................... ix  \nList of Tables ........................... xii  \nVita and Publications ....................... xiv  \nAbstract .............................. xv  \nChapter 1 Introduction ............................ 1  \n1.1 Introduction ......................... 1  \n1.1.1 Motivation: Why Context-Awareness Matters in Real-Time Audio .................. 1  \n1.1.2 Problem Statement & Thesis Objective ...... 1  \n1.1.3 Contributions Overview .............. 2  \n1.1.4 Fundamentals of Audio Scene Understanding ... 2  \n1.1.5 Causal & Non-Causal SED / TSE ......... 3  \n1.1.6 Research Gap .................... 3  \n1.1.7 Dissertation Roadmap ............... 4  \nChapter 2 Sound Event Detection (SED) in the Wild ........... 6  \n2.1 Overview ........................... 6  \n2.2 Introduction ......................... 8  \n2.3 Materials and Methods ................... 11  \n2.3.1 Field data collection ................ 11  \n2.3.2 Reference data collection .............. 11  \n2.3.3 Isolating vocalizations from xeno-canto recordings 17  \n2.3.4 Mel–spectrogram representations ......... 17  \n2.3.5 CNN multi-species classification models ...... 18  \n2.3.6 Inference on whole recording archive ....... 18  \n2.3.7 Evaluation of accuracy ............... 18  \n2.3.8 Post-classification corrections ........... 19  \n2.3.9 Comparison with BirdNET ............. 19  \n2.4 Results ............................ 20  \n2.4.1 Improvements in accuracy with pre-training of the CNN models with xeno-canto ........... 20  \n2.4.2 Differences between ROI and soundscape accuracy with CNN-XC Models ............. 21  \n2.4.3 Accuracy of post-classification corrections ..... 21  \n2.4.4 Comparison with BirdNET ............. 23  \n2.4.5 Optimal model selection for each species ..... 23  \n2.5 Discussion .......................... 27  \n2.5.1 Overall performance of bird vocalization models . 27  \n2.5.2 Mixture of Experts (MoE) ............. 29  \n2.5.3 Challenges of applied species-level bird classification 29  \n2.6 Conclusions ......................... 32  \nChapter 3 Quantifying Anthropophony, Geophony, Biophony ....... 34  \n3.1 Overview ........................... 34  \n3.2 Introduction ......................... 35  \n3.3 Methods ........................... 38  \n3.3.1 Study region and acoustic data collection ..... 38  \n3.3.2 Deep learning classification of soundscape components ......................... 40  \n3.3.3 Training dataset collection ............. 40  \n3.3.4 Spectrogram generation and cross-validation ... 41  \n3.3.5 CNN transfer learning ............... 41  \n3.3.6 Sound pattern statistical analyses ......... 4","cbCaiovrCdPZPO13","https://ap.wps.com/l/cbCaiovrCdPZPO13","pdf",13200328,1,162,"English","en",105,"# Chapter 1 Introduction\n## Motivation: Why Context-Awareness Matters in Real-Time Audio\n## Problem Statement & Thesis Objective\n## Contributions Overview\n## Fundamentals of Audio Scene Understanding\n## Causal & Non-Causal SED / TSE\n## Research Gap\n## Dissertation Roadmap\n# Chapter 2 Sound Event Detection (SED) in the Wild\n## Overview\n## Materials and Methods\n## Results\n## Discussion\n## Conclusions\n# Chapter 3 Quantifying Anthropophony, Geophony, Biophony\n## Overview\n## Methods\n## Results\n## Discussion\n## Conclusions\n# Chapter 4 Multi-task Target Sound Extraction (TSE)\n## Overview\n## Introduction\n## Method\n## Experiments\n## Discussion\n## Conclusion\n# Chapter 5 Reverberant Target Sound Extraction\n## Overview\n## Introduction\n## Related Work\n## Method\n## Experiments\n## Conclusion\n# Chapter 6 Real-time Causal Target Sound Extraction\n## Overview\n## Introduction\n## Related Work","[{\"question\":\"What is the dissertation’s main focus in context-aware real-time audio machine learning?\",\"answer\":\"The dissertation focuses on enabling sound understanding in real time using context-aware machine learning, covering sound event detection, soundscape component quantification, and target sound extraction under different constraints.\"},{\"question\":\"How does the work approach sound event detection in real-world recordings?\",\"answer\":\"It uses field and reference data collection, mel-spectrogram representations, CNN-based multi-species classification, inference over recording archives, post-classification corrections, and evaluation with comparisons to BirdNET.\"},{\"question\":\"What target sound extraction scenarios are covered?\",\"answer\":\"The dissertation covers multi-task target sound extraction, reverberant target sound extraction, and real-time causal target sound extraction, including model design, training, experiments, and ablation studies.\"}]","Context-Aware Real-Time Audio Machine Learning - Dissertation | PDF",1785898527,408,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":13,"keywords":34,"description":14,"schema_data":35,"social_meta":86,"head_meta":88,"extra_data":90,"updated_unix":28},"context-aware-real-time-audio-machine-learning-dissertation","",{"@graph":36,"@context":85},[37,54,68],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":20},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/document/","Document",2,{"item":49,"name":12,"@type":43,"position":50},"https://docshare.wps.com/document/research-report/",3,{"item":52,"name":13,"@type":43,"position":53},"https://docshare.wps.com/document/context-aware-real-time-audio-machine-learning-dissertation/125373/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":23,"description":14,"dateModified":62,"datePublished":62,"encodingFormat":61,"isAccessibleForFree":63,"interactionStatistic":64},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-08-05",true,{"@type":65,"interactionType":66,"userInteractionCount":4},"InteractionCounter",{"@type":67},"ViewAction",{"@type":69,"mainEntity":70},"FAQPage",[71,77,81],{"name":72,"@type":73,"acceptedAnswer":74},"What is the dissertation’s main focus in context-aware real-time audio machine learning?","Question",{"text":75,"@type":76},"The dissertation focuses on enabling sound understanding in real time using context-aware machine learning, covering sound event detection, soundscape component quantification, and target sound extraction under different constraints.","Answer",{"name":78,"@type":73,"acceptedAnswer":79},"How does the work approach sound event detection in real-world recordings?",{"text":80,"@type":76},"It uses field and reference data collection, mel-spectrogram representations, CNN-based multi-species classification, inference over recording archives, post-classification corrections, and evaluation with comparisons to BirdNET.",{"name":82,"@type":73,"acceptedAnswer":83},"What target sound extraction scenarios are covered?",{"text":84,"@type":76},"The dissertation covers multi-task target sound extraction, reverberant target sound extraction, and real-time causal target sound extraction, including model design, training, experiments, and ablation studies.","https://schema.org",{"og:url":52,"og:type":87,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":89,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":92},[93,97,101,105,110,115,120,123,128,131,135],{"id":20,"doc_module":4,"doc_module_name":46,"category_name":94,"show_sort_weight":95,"slug":96},"Story & Novel",90,"story-novel",{"id":47,"doc_module":4,"doc_module_name":46,"category_name":98,"show_sort_weight":99,"slug":100},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":46,"category_name":102,"show_sort_weight":103,"slug":104},"Exam",70,"exam",{"id":106,"doc_module":4,"doc_module_name":46,"category_name":107,"show_sort_weight":108,"slug":109},5,"Comic",60,"comic",{"id":111,"doc_module":4,"doc_module_name":46,"category_name":112,"show_sort_weight":113,"slug":114},6,"Technology",50,"technology",{"id":116,"doc_module":4,"doc_module_name":46,"category_name":117,"show_sort_weight":118,"slug":119},7,"Healthcare",40,"healthcare",{"id":11,"doc_module":4,"doc_module_name":46,"category_name":12,"show_sort_weight":121,"slug":122},30,"research-report",{"id":124,"doc_module":4,"doc_module_name":46,"category_name":125,"show_sort_weight":126,"slug":127},9,"Religion & Spirituality",20,"religion-spirituality",{"id":126,"doc_module":4,"doc_module_name":46,"category_name":129,"show_sort_weight":126,"slug":130},"World Cup","world-cup",{"id":132,"doc_module":4,"doc_module_name":46,"category_name":133,"show_sort_weight":132,"slug":134},10,"Lifestyle","lifestyle",{"id":136,"doc_module":4,"doc_module_name":46,"category_name":137,"show_sort_weight":106,"slug":138},19,"General","general"]