[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-241015-105":3,"detail-sidebar-cat-1-en-105":80,"doc-detail-241015-en":126},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":73,"head_meta":75,"extra_data":77,"updated_unix":79},105,"en","document-metadata-extraction-multimodal-241015","Document Metadata Extraction (Multimodal)","","This document outlines critical rules and specifications for extracting metadata from various document types, including text and images. It details a systematic approach to title generation, language detection, abstract creation, keyword identification, table of contents structuring, and FAQ generation. The instructions emphasize adherence to strict formatting requirements, including the output of a single, raw JSON object, and the mandatory inclusion of all eight specified fields: title, title_en, language, abstract, keywords, category, toc, and faqs. It prioritizes readable text from images and combines it with document text for comprehensive analysis. The system employs a hierarchical logic for information extraction and ensures language consistency across all generated fields, defaulting to English when no other language is confidently detected.",{"@graph":14,"@context":72},[15,34,55],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/template/","Template",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/template/general/","General",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/template/document-metadata-extraction-multimodal-241015/241015/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/document-metadata-extraction-multimodal-241015/241015.png","ImageObject",442,249,{"name":42,"@type":43},"Lucas Vance","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-25","2026-09-11",true,{"@type":52,"interactionType":53,"userInteractionCount":33},"InteractionCounter",{"@type":54},"ViewAction",{"@type":56,"mainEntity":57},"FAQPage",[58,64,68],{"name":59,"@type":60,"acceptedAnswer":61},"What is the primary goal of this document?","Question",{"text":62,"@type":63},"The primary goal is to provide a comprehensive and strict set of instructions for extracting structured metadata from documents, accommodating both text and image content.","Answer",{"name":65,"@type":60,"acceptedAnswer":66},"What programming format is required for the output?",{"text":67,"@type":63},"The response MUST be a single raw JSON object, starting with `{` and ending with `}` with no other characters before or after.",{"name":69,"@type":60,"acceptedAnswer":70},"How should the dominant language of a document be determined?",{"text":71,"@type":63},"The dominant language is determined by the majority of full natural language sentences. Isolated words, code, URLs, and numbers are ignored. If 80% or more of the sentences are in one language, that language is chosen.","https://schema.org",{"og:url":32,"og:type":74,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":76,"canonical":32},"index,follow",{"doc_id":78,"site_id":7},241015,1790032808,{"code":4,"msg":81,"data":82},"success",[83,88,93,98,103,108,113,118,123],{"id":84,"doc_module":22,"doc_module_name":25,"category_name":85,"show_sort_weight":86,"slug":87},11,"Presentations",90,"presentations",{"id":89,"doc_module":22,"doc_module_name":25,"category_name":90,"show_sort_weight":91,"slug":92},12,"Resumes",80,"resumes",{"id":94,"doc_module":22,"doc_module_name":25,"category_name":95,"show_sort_weight":96,"slug":97},14,"Invoices",70,"invoices",{"id":99,"doc_module":22,"doc_module_name":25,"category_name":100,"show_sort_weight":101,"slug":102},15,"Posters",60,"posters",{"id":104,"doc_module":22,"doc_module_name":25,"category_name":105,"show_sort_weight":106,"slug":107},16,"Social Media",50,"social-media",{"id":109,"doc_module":22,"doc_module_name":25,"category_name":110,"show_sort_weight":111,"slug":112},17,"Forms",40,"forms",{"id":114,"doc_module":22,"doc_module_name":25,"category_name":115,"show_sort_weight":116,"slug":117},18,"Letters",30,"letters",{"id":119,"doc_module":22,"doc_module_name":25,"category_name":120,"show_sort_weight":121,"slug":122},21,"Paper Templates",5,"papers-templates",{"id":124,"doc_module":22,"doc_module_name":25,"category_name":29,"show_sort_weight":4,"slug":125},158,"general-158",{"code":4,"msg":81,"data":127},{"doc_id":78,"user_id":128,"nickname":42,"user_avatar":129,"doc_module":22,"category_id":124,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":11,"file_id":130,"file_url":131,"file_type":132,"file_size":133,"view_count":33,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":94,"language":134,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":135,"faqs":136,"seo_title":137,"seo_description":12,"update_tm":138,"read_time":121},549768064622,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","cbCaiaw1xax8wcAw","https://ap.wps.com/l/cbCaiaw1xax8wcAw","pdf",369976,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the primary goal of this document?\",\"answer\":\"The primary goal is to provide a comprehensive and strict set of instructions for extracting structured metadata from documents, accommodating both text and image content.\"},{\"question\":\"What programming format is required for the output?\",\"answer\":\"The response MUST be a single raw JSON object, starting with `{` and ending with `}` with no other characters before or after.\"},{\"question\":\"How should the dominant language of a document be determined?\",\"answer\":\"The dominant language is determined by the majority of full natural language sentences. Isolated words, code, URLs, and numbers are ignored. If 80% or more of the sentences are in one language, that language is chosen.\"}]","Document Metadata Extraction (Multimodal) | PDF",1789168930]