[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-139020-105":59,"doc-detail-139020-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","countgd-multi-modal-open-world-counting-neurips-2024-paper","COUNTGD - Multi-Modal Open-World Counting - NeurIPS 2024 Paper","","Open-vocabulary object counting is extended to improve both generality and accuracy in open-world images. The paper repurposes GroundingDINO for counting and adds modules that let users specify target objects using text descriptions, visual exemplars, or both. Multi-modal prompting increases counting precision and flexibility, supports restricting counts with combined cues, and enables a single-stage open-world counting model named COUNTGD. Experiments report strong benchmark improvements across text-only and text+exemplar settings.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/countgd-multi-modal-open-world-counting-neurips-2024-paper/139020/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/countgd-multi-modal-open-world-counting-neurips-2024-paper/139020.png","ImageObject",300,407,{"name":92,"@type":93},"Angel","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-18","2026-08-23",true,{"@type":102,"interactionType":103,"userInteractionCount":19},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is COUNTGD designed to improve in open-world object counting?","Question",{"text":112,"@type":113},"COUNTGD targets higher generality and accuracy for open-vocabulary object counting in images by enabling flexible prompt-based object specification at inference time.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does COUNTGD specify the target object for counting?",{"text":117,"@type":113},"It accepts either text descriptions, visual exemplars, or both together, treating visual exemplars as text tokens and fusing them with text through attention mechanisms.",{"name":119,"@type":110,"acceptedAnswer":120},"What are the reported performance benefits of using text and visual exemplars together?",{"text":121,"@type":113},"Using both modalities significantly improves state-of-the-art results across counting benchmarks, while text-only performance remains comparable to or better than previous text-only approaches.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},139020,1787494171,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":19,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},687207412472,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","COUNTGD: Multi-Modal Open-World Counting  \nNiki Amini-Naieni Tengda Han Andrew Zisserman  \nVisual Geometry Group (VGG)  \nUniversity of Oxford {nikian,htd,[az}@robots.ox.ac.uk](az}@robots.ox.ac.uk)  \nAbstract  \nThe goal of this paper is to improve the generality and accuracy of open-vocabulary object counting in images. To improve the generality, we repurpose an openvocabulary detection foundation model (GroundingDINO) for the counting task, and also extend its capabilities by introducing modules to enable specifying the target object to count by visual exemplars. In turn, these new capabilities – being able to specify the target object by multi-modalites (text and exemplars)– lead to an improvement in counting accuracy. We make three contributions: first, we introduce the first open-world counting model, COUNTGD, where the prompt can be specified by a text description or visual exemplars or both; second, we show that the performance of the model significantly improves the state of the art on multiple counting benchmarks – when using text only, COUNTGD is comparable to or outperforms all previous text-only works, and when using both text and visual exemplars, we outperform all previous models; third, we carry out a preliminary study into different interactions between the text and visual exemplar prompts, including the cases where they reinforce each other and where one restricts the other. The code and an app to test the model are available at [https://www.robots.ox.ac.uk/vgg/research/countgd/](https://www.robots.ox.ac.uk/vgg/research/countgd/) .  \nFigure 1: COUNTGD is capable of taking both visual exemplars and text prompts to produce highly accurate object counts (a), but also seamlessly supports counting with only text queries or only visual exemplars (b) . The multi-modal visual exemplar and text queries bring extra flexibility to the open-world counting task, such as using a short phrase (c), or adding additional constraints (the words ‘left’ or ‘right’) to select a sub-set of the objects (d) . These examples are taken from the FSC-147 [42] and CountBench [39] test sets. The visual exemplars are shown as yellow boxes. (d) visualizes the predicted confidence map of the model, where a high color intensity indicates a high level of confidence.  \n1 Introduction  \nOpen-world object counting methods aim to enumerate all the instances of any category of object inan image. The ‘open-world’ refers to the model’s ability to count objects beyond the set of categories seen at training, thus enabling the user to specify categories of interest at inference without the need for model retraining. Recent techniques allow the user to specify the target object with only visual exemplars – bounding boxes around a few example objects in the image – [32, 35], or only text  \n38th Conference on Neural Information Processing Systems (NeurIPS 2024) .  \ndescriptions [1, 22] . By accepting either visual exemplars or text as prompts, open-world object counting methods can adapt to the specific object at inference time. This enables these techniques to count arbitrary classes of objects as specified by the user.  \nMethods that use visual exemplars to specify the object currently significantly outperform text-based counting methods on multiple benchmarks. This is because visual exemplars provide more detailed information than text – it can take many words to precisely describe an object; and perhaps more importantly, they provide intrinsic information on the object’s appearance – because the exemplars are from the same image they already ‘factor in’ the viewpoint and lighting, variables that significantly affect the object’s appearance. However, while visual exemplar-based approaches are more accurate, they limit the capabilities and generality of the counting model.  \nIn this paper we introduce a counting model that is able to specify the target object using visual exemplars, a text description, or both together. The model, named COUNTGD, has superior","cbCaitLWcsSLvwzF","https://ap.wps.com/l/cbCaitLWcsSLvwzF","pdf",5522028,28,"English","# Abstract\n# 1 Introduction\n## Open-world object counting motivation\n## Limitations of text-only vs exemplar-based methods\n## COUNTGD overview and contributions\n## Model architecture and multi-modal prompt mechanism","[{\"question\":\"What is COUNTGD designed to improve in open-world object counting?\",\"answer\":\"COUNTGD targets higher generality and accuracy for open-vocabulary object counting in images by enabling flexible prompt-based object specification at inference time.\"},{\"question\":\"How does COUNTGD specify the target object for counting?\",\"answer\":\"It accepts either text descriptions, visual exemplars, or both together, treating visual exemplars as text tokens and fusing them with text through attention mechanisms.\"},{\"question\":\"What are the reported performance benefits of using text and visual exemplars together?\",\"answer\":\"Using both modalities significantly improves state-of-the-art results across counting benchmarks, while text-only performance remains comparable to or better than previous text-only approaches.\"}]","COUNTGD - Multi-Modal Open-World Counting - NeurIPS 2024 Paper | PDF",71]