[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-185114-en":3,"doc-seo-185114-105":31,"detail-sidebar-cat-0-en-105":92},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":20,"is_deleted":4,"is_public":21,"is_downloadable":21,"audit_status":21,"page_count":22,"language":23,"language_code":24,"site_id":25,"html_lang":24,"table_of_contents":26,"faqs":27,"seo_title":28,"seo_description":14,"update_tm":29,"read_time":30},185114,549758146520,"Patrick","https://ap-avatar.wpscdn.com/avatar/80002397d8c0411e94?_k=1775819394049821470",19,"General","Document Metadata Extraction (Multimodal)","This document focuses on metadata extraction, specifically for multimodal inputs containing both text and images, and adheres to strict JSON output format requirements. The primary objective is to analyze document content, including extracted text from images, to generate structured metadata. Key fields to be extracted include title, English title, language, abstract, keywords, category, table of contents, and frequently asked questions. The process emphasizes accurate language detection, adherence to specific formatting rules, and the generation of comprehensive and relevant metadata for efficient document organization and retrieval. The document also details critical rules for title generation, including handling orphan serials, file name analysis, and document title extraction, with a strict 100-character limit for titles.","| No | Sampel | Jumlah |\n| --- | --- | --- |\n| 1. | Ketua Penitia Program Prakerin | 1 |\n| 2. | Guru Pemonitoring | 5 |\n| 3. | Pembimbing Industri (wakil DU/DI) | 5 |\n| 4. | Peserta didik BC SMK Veteran 1 Sukoharjo | 5 |","cbCaicPcl9OdaggZ","https://ap.wps.com/l/cbCaicPcl9OdaggZ","pdf",346414,2,1,10,"English","en",105,"# Document Metadata Extraction (Multimodal)","[{\"question\":\"What is the main purpose of this document?\",\"answer\":\"The main purpose is to define and guide the process of extracting structured metadata from multimodal documents, combining text and image analysis, and outputting the results in a specific JSON format.\"},{\"question\":\"What are the critical rules for title generation?\",\"answer\":\"Rules include rescuing orphan serials, prioritizing file names and document titles, formatting titles with hyphens for multi-component structures, and ensuring a 100-character limit. Titles must be in the document's detected language, with `title_en` being the English version.\"},{\"question\":\"How is the document language determined?\",\"answer\":\"Language is detected from meaningful sentences in the body text and image text, with a priority given to body text, then image text. The dominant language is determined by the majority of full sentences, defaulting to English if no other language is confidently detected.\"}]","Document Metadata Extraction (Multimodal) | PDF",1788365793,15,{"code":4,"msg":32,"data":33},"ok",{"site_id":25,"language":24,"slug":34,"title":13,"keywords":35,"description":14,"schema_data":36,"social_meta":87,"head_meta":89,"extra_data":91,"updated_unix":29},"document-metadata-extraction-multimodal","",{"@graph":37,"@context":86},[38,54,69],{"@type":39,"itemListElement":40},"BreadcrumbList",[41,45,48,51],{"item":42,"name":43,"@type":44,"position":21},"https://docshare.wps.com","Home","ListItem",{"item":46,"name":47,"@type":44,"position":20},"https://docshare.wps.com/document/","Document",{"item":49,"name":12,"@type":44,"position":50},"https://docshare.wps.com/document/general/",3,{"item":52,"name":13,"@type":44,"position":53},"https://docshare.wps.com/document/document-metadata-extraction-multimodal/185114/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":24,"description":14,"dateModified":62,"datePublished":63,"encodingFormat":61,"isAccessibleForFree":64,"interactionStatistic":65},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":42,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-09-05","2026-09-02",true,{"@type":66,"interactionType":67,"userInteractionCount":20},"InteractionCounter",{"@type":68},"ViewAction",{"@type":70,"mainEntity":71},"FAQPage",[72,78,82],{"name":73,"@type":74,"acceptedAnswer":75},"What is the main purpose of this document?","Question",{"text":76,"@type":77},"The main purpose is to define and guide the process of extracting structured metadata from multimodal documents, combining text and image analysis, and outputting the results in a specific JSON format.","Answer",{"name":79,"@type":74,"acceptedAnswer":80},"What are the critical rules for title generation?",{"text":81,"@type":77},"Rules include rescuing orphan serials, prioritizing file names and document titles, formatting titles with hyphens for multi-component structures, and ensuring a 100-character limit. Titles must be in the document's detected language, with `title_en` being the English version.",{"name":83,"@type":74,"acceptedAnswer":84},"How is the document language determined?",{"text":85,"@type":77},"Language is detected from meaningful sentences in the body text and image text, with a priority given to body text, then image text. The dominant language is determined by the majority of full sentences, defaulting to English if no other language is confidently detected.","https://schema.org",{"og:url":52,"og:type":88,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":90,"canonical":52},"index,follow",{"doc_id":7,"site_id":25},{"code":4,"msg":5,"data":93},[94,98,102,106,111,116,121,126,131,134,137],{"id":21,"doc_module":4,"doc_module_name":47,"category_name":95,"show_sort_weight":96,"slug":97},"Story & Novel",90,"story-novel",{"id":20,"doc_module":4,"doc_module_name":47,"category_name":99,"show_sort_weight":100,"slug":101},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":47,"category_name":103,"show_sort_weight":104,"slug":105},"Exam",70,"exam",{"id":107,"doc_module":4,"doc_module_name":47,"category_name":108,"show_sort_weight":109,"slug":110},5,"Comic",60,"comic",{"id":112,"doc_module":4,"doc_module_name":47,"category_name":113,"show_sort_weight":114,"slug":115},6,"Technology",50,"technology",{"id":117,"doc_module":4,"doc_module_name":47,"category_name":118,"show_sort_weight":119,"slug":120},7,"Healthcare",40,"healthcare",{"id":122,"doc_module":4,"doc_module_name":47,"category_name":123,"show_sort_weight":124,"slug":125},8,"Research & Report",30,"research-report",{"id":127,"doc_module":4,"doc_module_name":47,"category_name":128,"show_sort_weight":129,"slug":130},9,"Religion & Spirituality",20,"religion-spirituality",{"id":129,"doc_module":4,"doc_module_name":47,"category_name":132,"show_sort_weight":129,"slug":133},"World Cup","world-cup",{"id":22,"doc_module":4,"doc_module_name":47,"category_name":135,"show_sort_weight":22,"slug":136},"Lifestyle","lifestyle",{"id":11,"doc_module":4,"doc_module_name":47,"category_name":12,"show_sort_weight":107,"slug":138},"general"]