[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-188818-en":3,"doc-seo-188818-105":30,"detail-sidebar-cat-1-en-105":91},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":11,"category_id":12,"category_name":13,"doc_title":14,"doc_description":15,"doc_content":16,"file_id":17,"file_url":18,"file_type":19,"file_size":20,"view_count":11,"is_deleted":4,"is_public":11,"is_downloadable":11,"audit_status":11,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":15,"update_tm":28,"read_time":29},188818,2336477405376,"Stanley","https://ap-avatar.wpscdn.com/davatar_29158cc5080c5b710cf443261637dec0",1,158,"General","Document Metadata Extraction","This document outlines a comprehensive system for Document Metadata Extraction, leveraging both text and image analysis in a multimodal approach. The system is designed to extract crucial information such as document titles, abstract summaries, relevant keywords, a structured table of contents, and frequently asked questions (FAQs). It emphasizes strict adherence to language detection rules, ensuring all extracted text fields consistently match the dominant language of the document, with the exception of the `title_en` field which is always in English. The process involves parsing document text and readable text from images, prioritizing body text and image content for language detection, followed by the document title and file name. Detailed specifications are provided for each metadata field, including generation rules, length constraints, and output formatting requirements. The system aims to produce high-quality, SEO-friendly metadata that accurately reflects the document's content while maintaining structural integrity and consistency across all generated fields.","","cbCaisvMWkuSi8g5","https://ap.wps.com/l/cbCaisvMWkuSi8g5","pdf",611543,54,"English","en",105,"# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the primary goal of the Document Metadata Extraction system?\",\"answer\":\"The primary goal is to extract crucial information such as titles, abstracts, keywords, tables of contents, and FAQs from documents using a multimodal approach that analyzes both text and images.\"},{\"question\":\"How is the dominant language of a document determined?\",\"answer\":\"The dominant language is determined by analyzing the majority of full natural language sentences in the document's body text and any extracted image text. Isolated words or short phrases are disregarded.\"},{\"question\":\"What are the length constraints for the generated metadata fields?\",\"answer\":\"The `abstract` field has a minimum length of 450 characters and a maximum of 500 characters. The final title, including `title_en`, must not exceed 100 characters.\"}]","Document Metadata Extraction | PDF",1788392615,19,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":14,"keywords":16,"description":15,"schema_data":34,"social_meta":86,"head_meta":88,"extra_data":90,"updated_unix":28},"document-metadata-extraction-188818",{"@graph":35,"@context":85},[36,53,68],{"@type":37,"itemListElement":38},"BreadcrumbList",[39,43,47,50],{"item":40,"name":41,"@type":42,"position":11},"https://docshare.wps.com","Home","ListItem",{"item":44,"name":45,"@type":42,"position":46},"https://docshare.wps.com/template/","Template",2,{"item":48,"name":13,"@type":42,"position":49},"https://docshare.wps.com/template/general/",3,{"item":51,"name":14,"@type":42,"position":52},"https://docshare.wps.com/template/document-metadata-extraction-188818/188818/",4,{"url":51,"name":14,"@type":54,"author":55,"headline":14,"publisher":57,"fileFormat":60,"inLanguage":23,"description":15,"dateModified":61,"datePublished":62,"encodingFormat":60,"isAccessibleForFree":63,"interactionStatistic":64},"DigitalDocument",{"name":9,"@type":56},"Person",{"url":40,"name":58,"@type":59},"DocShare","Organization","application/pdf","2026-09-06","2026-09-02",true,{"@type":65,"interactionType":66,"userInteractionCount":11},"InteractionCounter",{"@type":67},"ViewAction",{"@type":69,"mainEntity":70},"FAQPage",[71,77,81],{"name":72,"@type":73,"acceptedAnswer":74},"What is the primary goal of the Document Metadata Extraction system?","Question",{"text":75,"@type":76},"The primary goal is to extract crucial information such as titles, abstracts, keywords, tables of contents, and FAQs from documents using a multimodal approach that analyzes both text and images.","Answer",{"name":78,"@type":73,"acceptedAnswer":79},"How is the dominant language of a document determined?",{"text":80,"@type":76},"The dominant language is determined by analyzing the majority of full natural language sentences in the document's body text and any extracted image text. Isolated words or short phrases are disregarded.",{"name":82,"@type":73,"acceptedAnswer":83},"What are the length constraints for the generated metadata fields?",{"text":84,"@type":76},"The `abstract` field has a minimum length of 450 characters and a maximum of 500 characters. The final title, including `title_en`, must not exceed 100 characters.","https://schema.org",{"og:url":51,"og:type":87,"og:title":14,"og:site_name":58,"og:description":15},"article",{"robots":89,"canonical":51},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":92},[93,98,103,108,113,118,123,128,133],{"id":94,"doc_module":11,"doc_module_name":45,"category_name":95,"show_sort_weight":96,"slug":97},11,"Presentations",90,"presentations",{"id":99,"doc_module":11,"doc_module_name":45,"category_name":100,"show_sort_weight":101,"slug":102},12,"Resumes",80,"resumes",{"id":104,"doc_module":11,"doc_module_name":45,"category_name":105,"show_sort_weight":106,"slug":107},14,"Invoices",70,"invoices",{"id":109,"doc_module":11,"doc_module_name":45,"category_name":110,"show_sort_weight":111,"slug":112},15,"Posters",60,"posters",{"id":114,"doc_module":11,"doc_module_name":45,"category_name":115,"show_sort_weight":116,"slug":117},16,"Social Media",50,"social-media",{"id":119,"doc_module":11,"doc_module_name":45,"category_name":120,"show_sort_weight":121,"slug":122},17,"Forms",40,"forms",{"id":124,"doc_module":11,"doc_module_name":45,"category_name":125,"show_sort_weight":126,"slug":127},18,"Letters",30,"letters",{"id":129,"doc_module":11,"doc_module_name":45,"category_name":130,"show_sort_weight":131,"slug":132},21,"Paper Templates",5,"papers-templates",{"id":12,"doc_module":11,"doc_module_name":45,"category_name":13,"show_sort_weight":4,"slug":134},"general-158"]