[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-242027-105":53,"doc-detail-242027-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","document-metadata-extraction-multimodal-242027","Document Metadata Extraction (Multimodal)","","This document provides instructions and specifications for extracting metadata from documents, particularly focusing on multimodal content involving text and images. It outlines critical rules for JSON output generation, including language detection and consistency, title formatting, and the extraction of abstract, keywords, table of contents, and frequently asked questions. The document emphasizes strict adherence to JSON formatting, language matching across all fields, and content-based accuracy for all extracted metadata. It also details specific requirements for each metadata field, such as title structure, abstract length and content, keyword selection, TOC formatting, and FAQ generation, ensuring comprehensive and structured document analysis.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/document-metadata-extraction-multimodal-242027/242027/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/document-metadata-extraction-multimodal-242027/242027.png","ImageObject",442,249,{"name":88,"@type":89},"Jacob","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-20","2026-09-12",true,{"@type":98,"interactionType":99,"userInteractionCount":76},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What is the primary purpose of this document?","Question",{"text":108,"@type":109},"The primary purpose of this document is to provide a comprehensive set of instructions and rules for extracting metadata from documents, especially those containing both text and images.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"What are the critical output requirements for the metadata extraction process?",{"text":113,"@type":109},"The response must be a single raw JSON object, strictly adhering to JSON formatting rules. All text fields must be in the detected document language, and all required fields like title, abstract, keywords, category, table of contents, and FAQs must be present.",{"name":115,"@type":106,"acceptedAnswer":116},"How is the document's language determined for metadata extraction?",{"text":117,"@type":109},"The document's language is determined by analyzing meaningful natural language sentences found in the body text and images. The dominant language, based on the majority of full sentences, is selected, with a default to English if no confident result is obtained.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},242027,1789184765,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":60,"file_id":130,"file_url":131,"file_type":132,"file_size":133,"view_count":9,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":134,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":60,"faqs":136,"seo_title":137,"seo_description":61,"update_tm":125,"read_time":47},962084931830,"https://ap-avatar.wpscdn.com/davatar_a8503ba1806abce46bf441b54a3ca4cd","cbCaierzNQQm7UUS","https://ap.wps.com/l/cbCaierzNQQm7UUS","pdf",184142,13,"English","[{\"question\":\"What is the primary purpose of this document?\",\"answer\":\"The primary purpose of this document is to provide a comprehensive set of instructions and rules for extracting metadata from documents, especially those containing both text and images.\"},{\"question\":\"What are the critical output requirements for the metadata extraction process?\",\"answer\":\"The response must be a single raw JSON object, strictly adhering to JSON formatting rules. All text fields must be in the detected document language, and all required fields like title, abstract, keywords, category, table of contents, and FAQs must be present.\"},{\"question\":\"How is the document's language determined for metadata extraction?\",\"answer\":\"The document's language is determined by analyzing meaningful natural language sentences found in the body text and images. The dominant language, based on the majority of full sentences, is selected, with a default to English if no confident result is obtained.\"}]","Document Metadata Extraction (Multimodal) | PDF"]