[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-186234-en":3,"doc-seo-186234-105":30,"detail-sidebar-cat-0-en-105":92},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":20,"is_deleted":4,"is_public":20,"is_downloadable":20,"audit_status":20,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":14,"update_tm":28,"read_time":29},186234,2336475401981,"Chumphorn","https://ap-avatar.wpscdn.com/avatar/22000c94efd8d5204d?x-image-process=image/resize,m_fixed,w_180,h_180&k=1786935347598174694",8,"Research & Report","Document Metadata Extraction (Multimodal)","This document outlines a multimodal approach to metadata extraction, combining image analysis with text-based processing. It details a methodology for extracting information from various document formats, including images, texts, and potentially other media. The presented framework includes preprocessing steps such as file resizing, data cleaning, and feature extraction. A key component is the integration of a Mask R-CNN model for object detection and bounding box identification, crucial for accurately pinpointing relevant information within visual data. The process involves an iterative refinement cycle, where detected objects are classified and appropriate actions are taken, such as marking or cropping. The document emphasizes the accuracy of the system through performance metrics like True Positives (TP), False Positives (FP), False Negatives (FN), and True Negatives (TN), demonstrating high accuracy rates across different background conditions. This research contributes to advanced information retrieval and document analysis systems by enabling a more robust and comprehensive extraction of metadata.","| Predicted Value | Actual Values |  |\n| --- | --- | --- |\n|  | Positif | Negatif |\n| Positif | TP | FP |\n| Negatif | FN | TN |\n\n| | | |\n| --- | --- | --- |\n| (a) (b) (c) |  |  |\n| | | |\n| (d) (e) (f) |  |  |\n| | | |\n| (g) (h) (i) |  |  |\n\n| Dataset | Total | Benar | Salah | Tingkat Akurasi |  |\n| --- | --- | --- | --- | --- | --- |\n| LJK Background Terang | 50 | 50 | 0 | 100% |  |\n| LJK Background Agak Gelap | 50 | 50 | 0 | 100% |  |\n| LJK Background Gelap | 50 | 50 | 0 |  | 100% |","cbCaihr4s3sTmQKY","https://ap.wps.com/l/cbCaihr4s3sTmQKY","pdf",1963530,1,10,"English","en",105,"# Document Metadata Extraction (Multimodal)\n## Instructions\n### Title Rules\n### `title_en`\n### `language`\n### abstract\n### `keywords`\n### `toc`\n### `faqs`\n### `category`\n## Output Requirements (STRICT)","[{\"question\":\"What is the primary objective of the document?\",\"answer\":\"The primary objective is to outline a multimodal approach for document metadata extraction, integrating image and text analysis.\"},{\"question\":\"What model is used for object detection and bounding box extraction?\",\"answer\":\"The document utilizes the Mask R-CNN model for object detection and bounding box extraction.\"},{\"question\":\"How is the performance of the system evaluated?\",\"answer\":\"The system's performance is evaluated using metrics such as True Positives, False Positives, False Negatives, and True Negatives, with accuracy rates demonstrated across various background conditions.\"}]","Document Metadata Extraction (Multimodal) | PDF",1788371957,25,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":13,"keywords":34,"description":14,"schema_data":35,"social_meta":87,"head_meta":89,"extra_data":91,"updated_unix":28},"document-metadata-extraction-multimodal","",{"@graph":36,"@context":86},[37,54,69],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":20},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/document/","Document",2,{"item":49,"name":12,"@type":43,"position":50},"https://docshare.wps.com/document/research-report/",3,{"item":52,"name":13,"@type":43,"position":53},"https://docshare.wps.com/document/document-metadata-extraction-multimodal/186234/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":23,"description":14,"dateModified":62,"datePublished":63,"encodingFormat":61,"isAccessibleForFree":64,"interactionStatistic":65},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-09-04","2026-09-02",true,{"@type":66,"interactionType":67,"userInteractionCount":20},"InteractionCounter",{"@type":68},"ViewAction",{"@type":70,"mainEntity":71},"FAQPage",[72,78,82],{"name":73,"@type":74,"acceptedAnswer":75},"What is the primary objective of the document?","Question",{"text":76,"@type":77},"The primary objective is to outline a multimodal approach for document metadata extraction, integrating image and text analysis.","Answer",{"name":79,"@type":74,"acceptedAnswer":80},"What model is used for object detection and bounding box extraction?",{"text":81,"@type":77},"The document utilizes the Mask R-CNN model for object detection and bounding box extraction.",{"name":83,"@type":74,"acceptedAnswer":84},"How is the performance of the system evaluated?",{"text":85,"@type":77},"The system's performance is evaluated using metrics such as True Positives, False Positives, False Negatives, and True Negatives, with accuracy rates demonstrated across various background conditions.","https://schema.org",{"og:url":52,"og:type":88,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":90,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":93},[94,98,102,106,111,116,121,124,129,132,135],{"id":20,"doc_module":4,"doc_module_name":46,"category_name":95,"show_sort_weight":96,"slug":97},"Story & Novel",90,"story-novel",{"id":47,"doc_module":4,"doc_module_name":46,"category_name":99,"show_sort_weight":100,"slug":101},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":46,"category_name":103,"show_sort_weight":104,"slug":105},"Exam",70,"exam",{"id":107,"doc_module":4,"doc_module_name":46,"category_name":108,"show_sort_weight":109,"slug":110},5,"Comic",60,"comic",{"id":112,"doc_module":4,"doc_module_name":46,"category_name":113,"show_sort_weight":114,"slug":115},6,"Technology",50,"technology",{"id":117,"doc_module":4,"doc_module_name":46,"category_name":118,"show_sort_weight":119,"slug":120},7,"Healthcare",40,"healthcare",{"id":11,"doc_module":4,"doc_module_name":46,"category_name":12,"show_sort_weight":122,"slug":123},30,"research-report",{"id":125,"doc_module":4,"doc_module_name":46,"category_name":126,"show_sort_weight":127,"slug":128},9,"Religion & Spirituality",20,"religion-spirituality",{"id":127,"doc_module":4,"doc_module_name":46,"category_name":130,"show_sort_weight":127,"slug":131},"World Cup","world-cup",{"id":21,"doc_module":4,"doc_module_name":46,"category_name":133,"show_sort_weight":21,"slug":134},"Lifestyle","lifestyle",{"id":136,"doc_module":4,"doc_module_name":46,"category_name":137,"show_sort_weight":107,"slug":138},19,"General","general"]