[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-189158-105":53,"doc-detail-189158-en":127},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":120,"head_meta":122,"extra_data":124,"updated_unix":126},105,"en","document-metadata-extraction-multimodal","Document Metadata Extraction (Multimodal)","","This document defines critical rules and guidelines for metadata extraction from various document types, including text and images. It emphasizes the importance of accurate language detection, using a hierarchical approach for title extraction from file names, document titles, and content. The guidelines detail specific formatting requirements for metadata fields such as title, title_en, language, abstract, keywords, category, table of contents (toc), and frequently asked questions (faqs). It stresses the need for multilingual consistency, ensuring all text-based fields adhere to the detected document language. The document also specifies length constraints and content generation rules for each field, particularly for the abstract and faqs, to ensure clarity, accuracy, and SEO-friendliness. Emphasis is placed on extracting genuine content features and avoiding misinterpretation of numbers or identifiers.",{"@graph":63,"@context":119},[64,80,102],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/document-metadata-extraction-multimodal/189158/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/document-metadata-extraction-multimodal/189158.png","ImageObject",442,249,{"name":88,"@type":89},"Riley West","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-29","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":101},"InteractionCounter",{"@type":100},"ViewAction",9,{"@type":103,"mainEntity":104},"FAQPage",[105,111,115],{"name":106,"@type":107,"acceptedAnswer":108},"What is the primary purpose of this document?","Question",{"text":109,"@type":110},"The primary purpose of this document is to outline the specific rules and instructions for extracting structured metadata from documents, which may include both text and image content.","Answer",{"name":112,"@type":107,"acceptedAnswer":113},"How should the document language be determined?",{"text":114,"@type":110},"The document language should be detected by analyzing meaningful natural language sentences within the body text and extracted image text, prioritizing the language used in the majority of full sentences.",{"name":116,"@type":107,"acceptedAnswer":117},"What are the requirements for the `title` field?",{"text":118,"@type":110},"The `title` field must be in the document's detected original language, strictly follow formatting rules including hyphenation for multi-component titles, and adhere to a maximum length of 100 characters. English titles must be provided in `title_en`.","https://schema.org",{"og:url":78,"og:type":121,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":123,"canonical":78},"index,follow",{"doc_id":125,"site_id":56},189158,1788394744,{"code":4,"msg":5,"data":128},{"doc_id":125,"user_id":129,"nickname":88,"user_avatar":130,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":60,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":101,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":9,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":126,"read_time":4},1099523885074,"https://ap-avatar.wpscdn.com/davatar_9964176cb1d06d4a9deccf72a44ae3dc","cbCaij42dEUYm7Xy","https://ap.wps.com/l/cbCaij42dEUYm7Xy","pdf",242059,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs\n**File name:**\n**Document title:** \n**Category definitions:** \n**Document text:** \n**Document images:**","[{\"question\":\"What is the primary purpose of this document?\",\"answer\":\"The primary purpose of this document is to outline the specific rules and instructions for extracting structured metadata from documents, which may include both text and image content.\"},{\"question\":\"How should the document language be determined?\",\"answer\":\"The document language should be detected by analyzing meaningful natural language sentences within the body text and extracted image text, prioritizing the language used in the majority of full sentences.\"},{\"question\":\"What are the requirements for the `title` field?\",\"answer\":\"The `title` field must be in the document's detected original language, strictly follow formatting rules including hyphenation for multi-component titles, and adhere to a maximum length of 100 characters. English titles must be provided in `title_en`.\"}]","Document Metadata Extraction (Multimodal) | PDF"]