[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-241412-105":3,"detail-sidebar-cat-1-en-105":80,"doc-detail-241412-en":126},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":73,"head_meta":75,"extra_data":77,"updated_unix":79},105,"en","document-metadata-extraction-multimodal-241412","Document Metadata Extraction (Multimodal)","","This document outlines the process for critical document metadata extraction, particularly focusing on multimodal analysis involving both text and images. It details strict instructions for parsing image content, combining it with textual data, and generating structured metadata in a raw JSON object format. The metadata schema includes fields for title, English title, language, abstract, keywords, category, table of contents, and frequently asked questions. Specific rules are provided for language detection, title formatting with hyphenation and length constraints, and ensuring language consistency across all output fields. The document emphasizes a multi-stage title generation process prioritizing orphan serial rescue, file name, and document title, with a fallback to dropping the document if no reliable title can be extracted. Detailed guidelines are given for feature detection within titles, including recognizing serial identifiers and formatting them appropriately. Furthermore, the abstract generation mandates a direct, professional, and SEO-friendly description within a specific character count, adhering to a strict prohibition of meta-talk and a third-person objective tone. Keyword extraction is limited to five items, while the table of contents requires a two-level markdown format with inferred outlines if explicit headings are absent. The FAQ section requires a minimum of three diverse questions with concise, text-grounded answers. Category classification relies on predefined definitions, with a fallback to \"General\" if no confident match is found.",{"@graph":14,"@context":72},[15,34,55],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/template/","Template",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/template/general/","General",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/template/document-metadata-extraction-multimodal-241412/241412/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/document-metadata-extraction-multimodal-241412/241412.png","ImageObject",442,249,{"name":42,"@type":43},"8796093062539","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-25","2026-09-12",true,{"@type":52,"interactionType":53,"userInteractionCount":33},"InteractionCounter",{"@type":54},"ViewAction",{"@type":56,"mainEntity":57},"FAQPage",[58,64,68],{"name":59,"@type":60,"acceptedAnswer":61},"What is the primary goal of this document?","Question",{"text":62,"@type":63},"The primary goal is to extract structured metadata from documents, combining text and image analysis, and outputting it in a raw JSON format with specific field requirements.","Answer",{"name":65,"@type":60,"acceptedAnswer":66},"What are the critical rules for language detection and consistency?",{"text":67,"@type":63},"Language is detected from meaningful sentences, prioritizing body text and image text. All text fields, including title, abstract, keywords, TOC, and FAQs, must strictly adhere to the detected dominant language.",{"name":69,"@type":60,"acceptedAnswer":70},"How are titles generated and formatted according to the instructions?",{"text":71,"@type":63},"Titles are generated through a prioritized sequence (orphan serial rescue, file name, document title), followed by feature detection and specific formatting rules including hyphenation for multi-component titles and a strict 100-character limit.","https://schema.org",{"og:url":32,"og:type":74,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":76,"canonical":32},"index,follow",{"doc_id":78,"site_id":7},241412,1790012746,{"code":4,"msg":81,"data":82},"success",[83,88,93,98,103,108,113,118,123],{"id":84,"doc_module":22,"doc_module_name":25,"category_name":85,"show_sort_weight":86,"slug":87},11,"Presentations",90,"presentations",{"id":89,"doc_module":22,"doc_module_name":25,"category_name":90,"show_sort_weight":91,"slug":92},12,"Resumes",80,"resumes",{"id":94,"doc_module":22,"doc_module_name":25,"category_name":95,"show_sort_weight":96,"slug":97},14,"Invoices",70,"invoices",{"id":99,"doc_module":22,"doc_module_name":25,"category_name":100,"show_sort_weight":101,"slug":102},15,"Posters",60,"posters",{"id":104,"doc_module":22,"doc_module_name":25,"category_name":105,"show_sort_weight":106,"slug":107},16,"Social Media",50,"social-media",{"id":109,"doc_module":22,"doc_module_name":25,"category_name":110,"show_sort_weight":111,"slug":112},17,"Forms",40,"forms",{"id":114,"doc_module":22,"doc_module_name":25,"category_name":115,"show_sort_weight":116,"slug":117},18,"Letters",30,"letters",{"id":119,"doc_module":22,"doc_module_name":25,"category_name":120,"show_sort_weight":121,"slug":122},21,"Paper Templates",5,"papers-templates",{"id":124,"doc_module":22,"doc_module_name":25,"category_name":29,"show_sort_weight":4,"slug":125},158,"general-158",{"code":4,"msg":81,"data":127},{"doc_id":78,"user_id":128,"nickname":42,"user_avatar":11,"doc_module":22,"category_id":124,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":129,"file_id":130,"file_url":131,"file_type":132,"file_size":133,"view_count":33,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":99,"language":134,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":11,"faqs":135,"seo_title":136,"seo_description":12,"update_tm":137,"read_time":121},8796093062539,"# Click to prove\n\nyou're human","cbCaijhkFP4AXvo8","https://ap.wps.com/l/cbCaijhkFP4AXvo8","pdf",173475,"English","[{\"question\":\"What is the primary goal of this document?\",\"answer\":\"The primary goal is to extract structured metadata from documents, combining text and image analysis, and outputting it in a raw JSON format with specific field requirements.\"},{\"question\":\"What are the critical rules for language detection and consistency?\",\"answer\":\"Language is detected from meaningful sentences, prioritizing body text and image text. All text fields, including title, abstract, keywords, TOC, and FAQs, must strictly adhere to the detected dominant language.\"},{\"question\":\"How are titles generated and formatted according to the instructions?\",\"answer\":\"Titles are generated through a prioritized sequence (orphan serial rescue, file name, document title), followed by feature detection and specific formatting rules including hyphenation for multi-component titles and a strict 100-character limit.\"}]","Document Metadata Extraction (Multimodal) | PDF",1789175083]