[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-242182-105":53,"doc-detail-242182-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","auto-checking-speech-transcriptions-by-multiple-template-constrained-posterior","Auto-Checking Speech Transcriptions by Multiple Template Constrained Posterior","","Checking transcription errors in speech databases is crucial yet time-consuming, traditionally relying on intensive manual review. Template Constrained Posterior (TCP) previously automated error checking using a single context template, but it lacked robustness and still required parameter optimization with some development effort. This work introduces multiple template constrained posterior methods using hypothesis sifting across full and loosely defined contexts, enabling confidence estimation at different expected accuracies without development data. Joint verification improves confidence measurement and remains robust across speech databases.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/auto-checking-speech-transcriptions-by-multiple-template-constrained-posterior/242182/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/auto-checking-speech-transcriptions-by-multiple-template-constrained-posterior/242182.png","ImageObject",442,249,{"name":88,"@type":89},"McGucket","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-25","2026-09-12",true,{"@type":98,"interactionType":99,"userInteractionCount":76},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What problem does the document address in speech transcription databases?","Question",{"text":108,"@type":109},"It addresses the need to detect transcription errors in speech databases efficiently, avoiding labor-intensive manual checking before the data can be trusted.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"How does the proposed approach improve over single-template TCP?",{"text":113,"@type":109},"It uses multiple context templates to provide more robust checking and eliminates the need for development data and parameter optimization used by single-template methods.",{"name":115,"@type":106,"acceptedAnswer":116},"How is sentence-level error detection evaluated in the experiments?",{"text":117,"@type":109},"The method generates ranked sentence lists by error likelihood; experimental results report rapidly decreasing sentence error hit rates among top-ranked sentences for different speech databases.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},242182,1789969777,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":76,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":79,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":139,"read_time":73},1236954412713,"https://us-avatar.wpscdn.com/davatar_29158cc5080c5b710cf443261637dec0","Auto-Checking Speech Transcriptions by Multiple Template Constrained  \nPosterior  \nLijuan WANG 1, Shenghao QIN2, Frank SOONG 1  \n1 Microsoft Research Asia, Beijing, China  \n2 Microsoft Business Division, Beijing, China  \n{lijuanw, sqin, [frankkps}@microsoft.com](frankkps}@microsoft.com)  \nAbstract  \nChecking transcription errors in speech database is an important but tedious task that traditionally requires intensive manual labor. In [9], Template Constrained Posterior (TCP) was proposed to automate the checking process by screening potential erroneous sentences with a single context template. However, single template-based method is not robust and requires parameter optimization that still involves some manual work. In this work, we propose to use multiple templates which is more robust and requires no development data for parameter optimization. By using its multiple hypothesis sifting capabilities -- from well-defined, full context to loosely defined context like wild card, the confidence for a focus unit can be measured at different expected accuracy. The joint verification by multiple TCP improves measured confidence of each unit in the transcription and is robust across different speech databases. Experimental results show that the checking process automatically separates erroneous sentences from correct ones: the sentence error hit rate decrease rapidly in the sorted TCP values, from 59% to 7% for the Mexican Spanish database and from 63% to 11% for the American English database, among the top 10% sentences in the rank lists.  \nIndex Terms: template constrained posterior, database checking  \n1. Introduction  \nHuman-computer voice interaction via text-to-speech and speech recognition has been an intensive subject of research for many years. One significant issue in this field is that nearly all work must rely upon a well-annotated speech database. For example, text-to-speech synthesis relies upon the accuracy of annotated phonetic labels and corresponding contexts for selecting good acoustic units from a pre-recorded database. However, such a database must be thoroughly examined before it may be relied upon, in order to catch reading or pronunciation errors, transcription errors, incomplete pronunciation lists, and similar issues. Because of the importance and wide application of this issue, automated detection of error is highly desirable, as illustrated in Fig. 1. Confidence is a useful measure for verifying speech transcription by assessing the reliability of a focused unit, such as a word, syllable, or phone.  \nA number of approaches for measuring confidence of speech transcriptions have been investigated [1-6] . They can be roughly classified into three major categories: Feature based approaches that attempt to assess confidence based on selected features, such as word duration, part of speech, word graph density, or using trained classifiers; 2) Explicit model based approaches that use a candidate class model with competing models, and a likelihood ratio test; 3) Posterior probability approaches that attempt to estimate the posterior  \nFigure 1: Illustration of auto-checking speech database  \nprobability of a recognized entity, given all acoustic observations. In our previous work [9-10], Template Constrained Posterior (TCP) was proposed for verifying transcription errors. A single context template is constructed to compute phone level TCP, which considers not only the focused phone, but also the partially matched contexts before and after the focused phone. However, single template-based method is not robust and requires parameter optimization (including context window length, partial matching ratio, KLD threshold for selecting confusable phones, and verification threshold) that still involves some manual work.  \nIn this work, we propose multiple template-based automatic checking which is more robust than our previous single template based approach and requires no development data for parameter optimization. These","cbCaiq3mNlkhc5Hl","https://ap.wps.com/l/cbCaiq3mNlkhc5Hl","pdf",317432,"English","# Introduction\n## Confidence for speech transcription checking\n# Template Constrained Posterior\n## From GPP to TCP\n# Auto-checking with multiple TCP\n## Procedure and joint verification\n# Experimental results\n## Sentence error separation\n# Conclusions","[{\"question\":\"What problem does the document address in speech transcription databases?\",\"answer\":\"It addresses the need to detect transcription errors in speech databases efficiently, avoiding labor-intensive manual checking before the data can be trusted.\"},{\"question\":\"How does the proposed approach improve over single-template TCP?\",\"answer\":\"It uses multiple context templates to provide more robust checking and eliminates the need for development data and parameter optimization used by single-template methods.\"},{\"question\":\"How is sentence-level error detection evaluated in the experiments?\",\"answer\":\"The method generates ranked sentence lists by error likelihood; experimental results report rapidly decreasing sentence error hit rates among top-ranked sentences for different speech databases.\"}]","Auto-Checking Speech Transcriptions by Multiple Template Constrained Posterior | PDF",1789187495]