[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-137716-105":59,"doc-detail-137716-en":134},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":127,"head_meta":129,"extra_data":131,"updated_unix":133},105,"en","knowledge-augmented-few-shot-visual-relation-detection-arxiv-230305342v1","Knowledge-augmented Few-shot Visual Relation Detection - arXiv 2303.05342v1","","Visual Relation Detection (VRD) seeks to recognize Subject–Predicate–Object relationships to support image understanding. Existing VRD systems often require thousands of labeled samples per relationship, while few-shot approaches still suffer from limited generalization due to the large semantic diversity and compositional variability of visual relations. The proposed knowledge-augmented few-shot VRD framework integrates textual knowledge from a pre-trained language model and visual relation knowledge from an automatically built knowledge graph, improving coverage and transfer. Experiments on three Visual Genome benchmarks show substantial gains over state-of-the-art models.",{"@graph":69,"@context":126},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/knowledge-augmented-few-shot-visual-relation-detection-arxiv-230305342v1/137716/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/knowledge-augmented-few-shot-visual-relation-detection-arxiv-230305342v1/137716.png","ImageObject",300,407,{"name":92,"@type":93},"Rizky","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-20","2026-08-22",true,{"@type":102,"interactionType":103,"userInteractionCount":44},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118,122],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does Visual Relation Detection (VRD) address?","Question",{"text":112,"@type":113},"VRD detects relationships in images in the form of Subject–Predicate–Object triplets, enabling deeper semantic understanding of visual interactions.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"Why do existing few-shot VRD models generalize poorly?",{"text":117,"@type":113},"They are constrained by the poor generalization capability caused by the vast semantic diversity and compositional nature of visual relationships.",{"name":119,"@type":110,"acceptedAnswer":120},"How does the proposed framework improve few-shot VRD generalization?",{"text":121,"@type":113},"It augments few-shot VRD with textual knowledge from a pre-trained language model and visual relation knowledge learned from an automatically constructed visual relation knowledge graph.",{"name":123,"@type":110,"acceptedAnswer":124},"What evidence is provided to validate the effectiveness of the framework?",{"text":125,"@type":113},"Experiments on three benchmarks from the Visual Genome dataset show performance surpassing existing state-of-the-art models with a large improvement.","https://schema.org",{"og:url":83,"og:type":128,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":130,"canonical":83},"index,follow",{"doc_id":132,"site_id":62},137716,1787438918,{"code":4,"msg":5,"data":135},{"doc_id":132,"user_id":136,"nickname":92,"user_avatar":137,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":138,"file_id":139,"file_url":140,"file_type":141,"file_size":142,"view_count":44,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":143,"language":144,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":145,"faqs":146,"seo_title":147,"seo_description":67,"update_tm":133,"read_time":148},962085564807,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","arXiv :2303 .05342v1 [ cs .CV] 9 Mar 2023  \nKnowledge-augmented Few-shot Visual Relation Detection  \nTianyu Yu 1 , Yangning Li 1 , Jiaoyan Chen2 , Yinghui Li 1 , Hai-Tao Zhengy1, Xi Cheny3, Qingbin Liu3 , Wenqiang Liu3 , Dongxiao Huang3 , Bei Wu3 , and Yexin Wang3  \n1 Shenzhen International Graduate School, Tsinghua University, Shenzhen, China  \n2Department of Computer Science, The University of Manchester, Mancester, UK  \n3Tencent, Shenzhen, China  \n[yiranytianyu@gmail.com](yiranytianyu@gmail.com)  \nAbstract  \nVisual Relation Detection (VRD) aims to detect relationships between objects for image understanding. Most existing VRD methods rely on thousands of training samples of each relationship to achieve satisfactory performance. Some recent papers tackle this problem by few-shot learning with elaborately designed pipelines and pre-trained word vectors. However, the performance of existing few-shot VRD models is severely hampered by the poor generalization capability, as they struggle to handle the vast semantic diversity of visual relationships. Nonetheless, humans have the ability to learn new relationships with just few examples based on their knowledge. Inspired by this, we devisea knowledge-augmented, few-shot VRD framework leveraging both textual knowledge and visual relation knowledge to improve the generalization ability of few-shot VRD. The textual knowledge and visual relation knowledge are acquired from a pre-trained language model and an automatically constructed visual relation knowledge graph, respectively. We extensively validate the effectiveness of our framework. Experiments conducted on three benchmarks from the commonly used Visual Genome dataset show that our performance surpasses existing state-of-the-art models with a large improvement.  \n1. Introduction  \nVisual Relation Detection (VRD) targets at the detection of relations, i.e. Subject-Predicate-Object triplets, which capture a wide variety of interactions (relationships) between objects in images. For example, the input image in Figure 1 shows the  eating relationship between the  \ny Corresponding authors: [zheng.haitao@sz.tsinghua.edu.cn](zheng.haitao@sz.tsinghua.edu.cn), jasonx[chen@tencent.com](chen@tencent.com)  \nFigure 1: Leveraging external knowledge to augment fewshot VRD. Due to the large semantic diversity of eating, a model trained with the few-shot examples, such as horse-eating-grass or child-eating-hotdog cannot be easily generalized to new examples, such as dog-eating-apple. While external knowledge can be leveraged to improve the model's generalization ability. For instance, textual knowledge from large textual corpora can describe meanings of dog, apple, and eating. Visual relation knowledge learnt from the visual relation knowledge graph can show common relationships between objects, such as the fact that eating is a common relationship between home pets (e.g., dog, rabbits) and fruits (e.g., apple) .  \ndog and the apple. The detected relations provide a deep and comprehensive understanding of the semantic content of images and have facilitated state-of-the-art models in numerous applications in computer vision such as visual question answering[18, 39], image retrieval [19], and image captioning [45, 23] .  \nMost existing VRD methods [29, 22, 10] rely on a large amount of training data for each relationship to achieve satisfactory performance. However, the distribution of relationships in real-world images is extremely long-tail since many relationships are inherently uncommon. Speciﬁcally, 92.3% of relationships in the Visual Genome dataset [21] have only ten or fewer samples (e.g., packing, spouting), while wearing has about 12,400 times more samples. As a result, these methods can only handle those frequent relationships which have many labeled samples. In order to enable VRD with limited samples, some recent works try to solve this task in the few-shot setting with either visual relation detection pre-training on frequent relatio","cbCaiiSNOAZg4cgg","https://ap.wps.com/l/cbCaiiSNOAZg4cgg","pdf",1729492,11,"English","# Abstract\n# Introduction\n## Visual Relation Detection and the Few-shot Challenge\n## Knowledge-augmented Few-shot VRD Framework\n### Textual Knowledge\n### Visual Relation Knowledge","[{\"question\":\"What problem does Visual Relation Detection (VRD) address?\",\"answer\":\"VRD detects relationships in images in the form of Subject–Predicate–Object triplets, enabling deeper semantic understanding of visual interactions.\"},{\"question\":\"Why do existing few-shot VRD models generalize poorly?\",\"answer\":\"They are constrained by the poor generalization capability caused by the vast semantic diversity and compositional nature of visual relationships.\"},{\"question\":\"How does the proposed framework improve few-shot VRD generalization?\",\"answer\":\"It augments few-shot VRD with textual knowledge from a pre-trained language model and visual relation knowledge learned from an automatically constructed visual relation knowledge graph.\"},{\"question\":\"What evidence is provided to validate the effectiveness of the framework?\",\"answer\":\"Experiments on three benchmarks from the Visual Genome dataset show performance surpassing existing state-of-the-art models with a large improvement.\"}]","Knowledge-augmented Few-shot Visual Relation Detection - arXiv 2303.05342v1 | PDF",28]