[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-197817-105":53,"doc-detail-197817-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","document-metadata-extraction-multimodal-197817","Document Metadata Extraction (Multimodal)","","This document details a multimodal approach to metadata extraction. It presents a method for analyzing and understanding information presented in both text and image formats, utilizing advanced AI techniques. The core of the document showcases visual explanations of how the AI model identifies salient features within images, illustrated through heatmaps that highlight areas of interest relevant to the task. Comparisons are made between the original images, a method labeled \"\u003Cbecause>\", and the proposed \"Ours\" method, demonstrating the effectiveness and interpretability of the AI's analysis. The document also includes technical specifications for metadata generation, outlining requirements for title, language, abstract, keywords, table of contents, and frequently asked questions. The emphasis is on structured, machine-readable output using JSON format, adherence to language consistency rules, and the extraction of comprehensive metadata to enhance document discoverability and organization through AI.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/document-metadata-extraction-multimodal-197817/197817/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/document-metadata-extraction-multimodal-197817/197817.png","ImageObject",442,249,{"name":88,"@type":89},"Tawan","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-26","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":79},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What is the primary purpose of this document?","Question",{"text":108,"@type":109},"This document describes a multimodal method for metadata extraction, focusing on an AI's ability to understand and represent information from both text and images, particularly using heatmaps to visualize analysis focus.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"What are the key components of the proposed \"Ours\" method presented in the images?",{"text":113,"@type":109},"The \"Ours\" method, as shown in the comparisons, generates visual explanations (heatmaps) on original images to highlight the AI's focus areas during its analysis, offering a more interpretable output than the \"\u003Cbecause>\" method.",{"name":115,"@type":106,"acceptedAnswer":116},"What are the general requirements for the output JSON object?",{"text":117,"@type":109},"The output must be a single raw JSON object containing specific fields like title, language, abstract, keywords, category, table of contents, and FAQs, all strictly formatted and adhering to specified content and language rules.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},197817,1790244545,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":60,"file_id":130,"file_url":131,"file_type":132,"file_size":133,"view_count":79,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":9,"language":134,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":135,"faqs":136,"seo_title":137,"seo_description":61,"update_tm":138,"read_time":4},2336475104042,"https://ap-avatar.wpscdn.com/avatar/22000c4c32af1715be0?x-image-process=image/resize,m_fixed,w_180,h_180&k=1786537525561427321","cbCaicdWkjvWVLQf","https://ap.wps.com/l/cbCaicdWkjvWVLQf","pdf",1131474,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the primary purpose of this document?\",\"answer\":\"This document describes a multimodal method for metadata extraction, focusing on an AI's ability to understand and represent information from both text and images, particularly using heatmaps to visualize analysis focus.\"},{\"question\":\"What are the key components of the proposed \\\"Ours\\\" method presented in the images?\",\"answer\":\"The \\\"Ours\\\" method, as shown in the comparisons, generates visual explanations (heatmaps) on original images to highlight the AI's focus areas during its analysis, offering a more interpretable output than the \\\"\\u003cbecause\\u003e\\\" method.\"},{\"question\":\"What are the general requirements for the output JSON object?\",\"answer\":\"The output must be a single raw JSON object containing specific fields like title, language, abstract, keywords, category, table of contents, and FAQs, all strictly formatted and adhering to specified content and language rules.\"}]","Document Metadata Extraction (Multimodal) | PDF",1788471020]