[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-237797-105":53,"doc-detail-237797-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","document-metadata-extraction-237797","Document Metadata Extraction","","This document outlines the critical process of Document Metadata Extraction, emphasizing a multimodal approach that integrates both text and image data. It details strict rules for language detection and consistency across all metadata fields, including title, abstract, keywords, table of contents, and FAQs. The document specifies precise formatting requirements for JSON output, ensuring valid structure and correct escaping of special characters. It also mandates a rigorous self-check process for language, content, and formatting accuracy before final output, with a clear hierarchy for title generation from file names, document content, or inferential methods, and includes specific guidelines for feature detection and title formatting to ensure optimal SEO and readability.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/document-metadata-extraction-237797/237797/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/document-metadata-extraction-237797/237797.png","ImageObject",442,249,{"name":88,"@type":89},"Berry Peter","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-26","2026-09-11",true,{"@type":98,"interactionType":99,"userInteractionCount":79},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"What is the primary goal of the Document Metadata Extraction process outlined?","Question",{"text":108,"@type":109},"The primary goal is to analyze document content, including text and images, and generate structured metadata in a precise JSON format, adhering to strict language and formatting rules.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"How is the dominant language of a document determined according to the rules?",{"text":113,"@type":109},"The dominant language is determined by the majority of full natural language sentences. Isolated words, code, URLs, numbers, and filler text are ignored in this determination.",{"name":115,"@type":106,"acceptedAnswer":116},"What are the constraints on the 'title' field in the generated metadata?",{"text":117,"@type":109},"The 'title' field must be in the document's detected original language, strictly adhere to structural separation rules (using hyphens), and not exceed 100 characters. Foreign elements within the title must be translated to the primary language.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},237797,1790001502,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":79,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":15,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":139,"read_time":79},1374402524268,"https://ap-avatar.wpscdn.com/davatar_3d24733baf745e90a7e4bdd5f77d97b2","# Click to verify","cbCaidu38SkJQ1eX","https://ap.wps.com/l/cbCaidu38SkJQ1eX","pdf",333030,"English","# Document Metadata Extraction (Multimodal)\n## Instructions\n## Language Rules\n## Field Specifications\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)\n## Inputs","[{\"question\":\"What is the primary goal of the Document Metadata Extraction process outlined?\",\"answer\":\"The primary goal is to analyze document content, including text and images, and generate structured metadata in a precise JSON format, adhering to strict language and formatting rules.\"},{\"question\":\"How is the dominant language of a document determined according to the rules?\",\"answer\":\"The dominant language is determined by the majority of full natural language sentences. Isolated words, code, URLs, numbers, and filler text are ignored in this determination.\"},{\"question\":\"What are the constraints on the 'title' field in the generated metadata?\",\"answer\":\"The 'title' field must be in the document's detected original language, strictly adhere to structural separation rules (using hyphens), and not exceed 100 characters. Foreign elements within the title must be translated to the primary language.\"}]","Document Metadata Extraction | PDF",1789125892]