[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-179866-105":3,"detail-sidebar-cat-0-en-105":80,"doc-detail-179866-en":130},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":73,"head_meta":75,"extra_data":77,"updated_unix":79},105,"en","document-metadata-extraction-179866","Document Metadata Extraction","","This document outlines critical instructions and rules for extracting metadata from document content, including text and images. It emphasizes accuracy in language detection, title generation, and the inclusion of essential fields such as abstract, keywords, table of contents, and frequently asked questions. The system must adhere to strict formatting requirements, outputting a single-line JSON object and ensuring all text fields are in the detected document language. Special attention is given to title formatting, including handling multi-component titles and feature identifiers, with a strict character limit. The document also details rules for abstract generation, keyword selection, table of contents structuring, and FAQ creation, promoting diversity and relevance in the generated metadata. Failure to comply with any of these directives will result in an invalid output.",{"@graph":14,"@context":72},[15,34,55],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/general/","General",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/document-metadata-extraction-179866/179866/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/document-metadata-extraction-179866/179866.png","ImageObject",300,407,{"name":42,"@type":43},"Chloe Bennett","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-19","2026-09-02",true,{"@type":52,"interactionType":53,"userInteractionCount":22},"InteractionCounter",{"@type":54},"ViewAction",{"@type":56,"mainEntity":57},"FAQPage",[58,64,68],{"name":59,"@type":60,"acceptedAnswer":61},"What is the primary output format required for this document analysis?","Question",{"text":62,"@type":63},"The response MUST be a single raw JSON object, starting with '{' and ending with '}', with no markdown or code fences.","Answer",{"name":65,"@type":60,"acceptedAnswer":66},"What is the critical language rule for the metadata fields?",{"text":67,"@type":63},"All text fields such as title, abstract, keywords, toc, and faqs MUST be written in the same language as the detected document language. The 'title_en' field is the only exception, always being in English.",{"name":69,"@type":60,"acceptedAnswer":70},"What is the maximum length for the final title?",{"text":71,"@type":63},"The final title MUST be 100 characters or fewer. If the generated title exceeds this limit, the action/style phrase should be truncated first, followed by shortening subtitles if necessary, while preserving the core title and serial numbers.","https://schema.org",{"og:url":32,"og:type":74,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":76,"canonical":32},"index,follow",{"doc_id":78,"site_id":7},179866,1788338553,{"code":4,"msg":81,"data":82},"success",[83,87,91,95,100,105,110,115,120,123,127],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":84,"show_sort_weight":85,"slug":86},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":88,"show_sort_weight":89,"slug":90},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":92,"show_sort_weight":93,"slug":94},"Exam",70,"exam",{"id":96,"doc_module":4,"doc_module_name":25,"category_name":97,"show_sort_weight":98,"slug":99},5,"Comic",60,"comic",{"id":101,"doc_module":4,"doc_module_name":25,"category_name":102,"show_sort_weight":103,"slug":104},6,"Technology",50,"technology",{"id":106,"doc_module":4,"doc_module_name":25,"category_name":107,"show_sort_weight":108,"slug":109},7,"Healthcare",40,"healthcare",{"id":111,"doc_module":4,"doc_module_name":25,"category_name":112,"show_sort_weight":113,"slug":114},8,"Research & Report",30,"research-report",{"id":116,"doc_module":4,"doc_module_name":25,"category_name":117,"show_sort_weight":118,"slug":119},9,"Religion & Spirituality",20,"religion-spirituality",{"id":118,"doc_module":4,"doc_module_name":25,"category_name":121,"show_sort_weight":118,"slug":122},"World Cup","world-cup",{"id":124,"doc_module":4,"doc_module_name":25,"category_name":125,"show_sort_weight":124,"slug":126},10,"Lifestyle","lifestyle",{"id":128,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":96,"slug":129},19,"general",{"code":4,"msg":81,"data":131},{"doc_id":78,"user_id":132,"nickname":42,"user_avatar":133,"doc_module":4,"category_id":128,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":11,"file_id":134,"file_url":135,"file_type":136,"file_size":137,"view_count":4,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":138,"language":139,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":12,"update_tm":79,"read_time":143},962084925782,"https://ap-avatar.wpscdn.com/davatar_9964176cb1d06d4a9deccf72a44ae3dc","cbCaiePBkL1tN3t9","https://ap.wps.com/l/cbCaiePBkL1tN3t9","pdf",1322695,16,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n### Language Rules\n### Field Specifications\n#### Title Rules\n#### `title_en`\n#### `language`\n#### abstract\n#### `keywords`\n#### `toc`\n#### `faqs`\n#### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the primary output format required for this document analysis?\",\"answer\":\"The response MUST be a single raw JSON object, starting with '{' and ending with '}', with no markdown or code fences.\"},{\"question\":\"What is the critical language rule for the metadata fields?\",\"answer\":\"All text fields such as title, abstract, keywords, toc, and faqs MUST be written in the same language as the detected document language. The 'title_en' field is the only exception, always being in English.\"},{\"question\":\"What is the maximum length for the final title?\",\"answer\":\"The final title MUST be 100 characters or fewer. If the generated title exceeds this limit, the action/style phrase should be truncated first, followed by shortening subtitles if necessary, while preserving the core title and serial numbers.\"}]","Document Metadata Extraction | PDF",25]