[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-197065-105":53,"doc-detail-197065-en":127},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":120,"head_meta":122,"extra_data":124,"updated_unix":126},105,"en","document-metadata-extraction-197065","Document Metadata Extraction","","This document focuses on the critical task of Document Metadata Extraction, specifically employing a multimodal approach that integrates both text and image analysis. The objective is to extract comprehensive and structured metadata from various document types. The process involves accurately identifying and extracting readable text from images, which is then combined with the existing document text to create a unified dataset for analysis. This integrated data is crucial for generating detailed metadata, including titles, abstracts, keywords, tables of contents, and frequently asked questions. The system emphasizes adherence to strict language rules, ensuring that all extracted and generated text fields maintain consistency with the document's dominant language. Furthermore, specialized rules are applied for title generation, including rescuing or inferring titles from filenames or content when a formal title is absent, and formatting them according to specific length and structural requirements. The system also dictates the categorization of documents based on predefined category definitions, with a fallback to 'General' if no specific category is a strong match.",{"@graph":63,"@context":119},[64,80,102],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/document-metadata-extraction-197065/197065/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/document-metadata-extraction-197065/197065.png","ImageObject",442,249,{"name":88,"@type":89},"\tCallum ","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-27","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":101},"InteractionCounter",{"@type":100},"ViewAction",7,{"@type":103,"mainEntity":104},"FAQPage",[105,111,115],{"name":106,"@type":107,"acceptedAnswer":108},"What is the main goal of this document?","Question",{"text":109,"@type":110},"The main goal is to extract structured metadata from documents using a multimodal approach that combines text and image analysis.","Answer",{"name":112,"@type":107,"acceptedAnswer":113},"What are the key steps involved in the metadata extraction process?",{"text":114,"@type":110},"The process involves extracting text from images, combining it with document text, and then generating metadata fields such as title, abstract, keywords, table of contents, and FAQs.",{"name":116,"@type":107,"acceptedAnswer":117},"What are the critical language rules that must be followed?",{"text":118,"@type":110},"All text fields must be in the same language as the detected dominant language of the document. This includes title, abstract, keywords, table of contents, and FAQs.","https://schema.org",{"og:url":78,"og:type":121,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":123,"canonical":78},"index,follow",{"doc_id":125,"site_id":56},197065,1788464697,{"code":4,"msg":5,"data":128},{"doc_id":125,"user_id":129,"nickname":88,"user_avatar":130,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":60,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":101,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":79,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":126,"read_time":73},137451211410,"https://ap-avatar.wpscdn.com/avatar/2000bb0a9246f588df?x-image-process=image/resize,m_fixed,w_180,h_180&k=1786362646172706240","cbCairbJs1taCatD","https://ap.wps.com/l/cbCairbJs1taCatD","pdf",858428,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the main goal of this document?\",\"answer\":\"The main goal is to extract structured metadata from documents using a multimodal approach that combines text and image analysis.\"},{\"question\":\"What are the key steps involved in the metadata extraction process?\",\"answer\":\"The process involves extracting text from images, combining it with document text, and then generating metadata fields such as title, abstract, keywords, table of contents, and FAQs.\"},{\"question\":\"What are the critical language rules that must be followed?\",\"answer\":\"All text fields must be in the same language as the detected dominant language of the document. This includes title, abstract, keywords, table of contents, and FAQs.\"}]","Document Metadata Extraction | PDF"]