[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-196545-105":53,"doc-detail-196545-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","multimodal-document-metadata-extraction-multimodal","Multimodal Document Metadata Extraction (Multimodal)","","This document outlines a multimodal approach to document metadata extraction, leveraging both text and image analysis. It details a system that integrates Whisper encoders with a Qwen model for processing audio and text data, featuring various encoder blocks, RoPE layers, and attention mechanisms. The system employs LoRA adaptation for trainable parameters and includes acoustic soft tokens and auxiliary heads for enhanced performance. The training objective is defined by a combination of Mean Squared Error (MSE) loss for both main and auxiliary outputs, with a specified tolerance. Various models, including CASA and CASA-Crisper, are benchmarked against others using metrics like RMSE, PCC, and classification scores (A2, B1, B2, C1, Macro). Parameter counts and confidence intervals are also presented, illustrating the system's performance and efficiency.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/multimodal-document-metadata-extraction-multimodal/196545/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/multimodal-document-metadata-extraction-multimodal/196545.png","ImageObject",442,249,{"name":88,"@type":89},"Evangeline","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-26","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":79},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What is the primary goal of the described system?","Question",{"text":108,"@type":109},"The primary goal is to enable multimodal document metadata extraction by integrating audio and text processing with AI models like Whisper and Qwen.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"What are the key components of the described system architecture?",{"text":113,"@type":109},"The system includes Whisper encoders and decoders, LoRA for trainable parameters, encoder blocks with RoPE, acoustic soft tokens, auxiliary heads, and a Qwen-2.5-2B model for processing fused input embeddings.",{"name":115,"@type":106,"acceptedAnswer":116},"How is the system's performance evaluated?",{"text":117,"@type":109},"The system's performance is evaluated using metrics such as Root Mean Squared Error (RMSE), Pearson Correlation Coefficient (PCC), and classification accuracy on specific tasks (A2, B1, B2, C1, Macro). Total parameter count is also considered.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},196545,1788458225,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":79,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":47,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":125,"read_time":73},13056703019662,"https://ap-avatar.wpscdn.com/avatar/be000253a8e92610077?_k=1778726343310543188","| Model | RMSE | PCC | %≤0.5 | %≤1.0 |\n| --- | --- | --- | --- | --- |\n| NTNU [3] | 0.360 | 0.827 | 85.7 | 99.0 |\n| Perezoso [5] | 0.364 | 0.826 | 83.0 | 99.7 |\n| CASA | 0.358 | 0.829 | 84.7 | 98.7 |\n| CASA-Crisper | 0.363 | 0.836 | 84.0 | 99.7 |\n\n\n| Model | RMSE | Total parameter |\n| --- | --- | --- |\n| CASA | 0.358 | 3.13 B |\n| NTNU [3] | 0.360 | 6.24 B |\n| Perezoso [5] | 0.364 | 2.17 B |\n| One Whisper [9] | 0.372 | 0.17 B |\n\n\n| Model | A2 | B1 | B2 | C1 | Macro |\n| --- | --- | --- | --- | --- | --- |\n| CASA | 0.553 | 0.351 | 0.290 | 0.554 | 0.437 |\n| CASA-Crisper | 0.485 | 0.335 | 0.322 | 0.617 | 0.440 |\n\n\n| Model | Mean | Median | Min–Max | 95% CI |\n| --- | --- | --- | --- | --- |\n| CASA | 0.363 | 0.362 | 0.357–0.377 | [0.359, 0.367] |\n| aux-0 | 0.367 | 0.366 | 0.362–0.376 | [0.364, 0.370] |\n| 4e-4 | 0.378 | 0.376 | 0.350–0.402 | [0.364, 0.392] |","cbCaiokMvXtUKulK","https://ap.wps.com/l/cbCaiokMvXtUKulK","pdf",823484,"English","# Model Comparison","[{\"question\":\"What is the primary goal of the described system?\",\"answer\":\"The primary goal is to enable multimodal document metadata extraction by integrating audio and text processing with AI models like Whisper and Qwen.\"},{\"question\":\"What are the key components of the described system architecture?\",\"answer\":\"The system includes Whisper encoders and decoders, LoRA for trainable parameters, encoder blocks with RoPE, acoustic soft tokens, auxiliary heads, and a Qwen-2.5-2B model for processing fused input embeddings.\"},{\"question\":\"How is the system's performance evaluated?\",\"answer\":\"The system's performance is evaluated using metrics such as Root Mean Squared Error (RMSE), Pearson Correlation Coefficient (PCC), and classification accuracy on specific tasks (A2, B1, B2, C1, Macro). Total parameter count is also considered.\"}]","Multimodal Document Metadata Extraction (Multimodal) | PDF"]