[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-156473-105":59,"doc-detail-156473-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","vicrit-a-verifiable-reinforcement-learning-proxy-task-for-visual-perception-in-vlms","ViCrit - A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs","","Reinforcement learning enables effective fine-tuning of large language models when training signals are both challenging and automatically verifiable. Extending this benefit to visual perception in vision–language models is hindered by the lack of vision-centric tasks that remain unambiguous to grade. ViCrit introduces an RL proxy task that injects a subtle, synthetic visual hallucination into human-written image captions and trains models to localize the corrupted span using an exact-match reward. ViCrit-Bench is proposed for systematic evaluation across domains and error types. Experiments show improved VL benchmark performance and strong transfer to abstract image reasoning and visual math.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/vicrit-a-verifiable-reinforcement-learning-proxy-task-for-visual-perception-in-vlms/156473/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/vicrit-a-verifiable-reinforcement-learning-proxy-task-for-visual-perception-in-vlms/156473.png","ImageObject",300,407,{"name":92,"@type":93},"Finn","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-20","2026-08-28",true,{"@type":102,"interactionType":103,"userInteractionCount":24},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does ViCrit address in visual perception for VLMs?","Question",{"text":112,"@type":113},"ViCrit targets the lack of vision-centric RL tasks that are both perceptually challenging and unambiguously verifiable, which limits progress in improving VLM visual perception.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does the ViCrit proxy task work?",{"text":117,"@type":113},"ViCrit starts from human-written image captions and injects a single subtle visual description error into the caption, then trains the model to locate the corrupted span given the image and modified caption.",{"name":119,"@type":110,"acceptedAnswer":120},"What is ViCrit-Bench and why is it introduced?",{"text":121,"@type":113},"ViCrit-Bench is a category-balanced diagnostic benchmark that systematically probes perception errors across diverse image domains and different error types to facilitate evaluation.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},156473,1787959541,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":24,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},34359740700684,"https://ap-avatar.wpscdn.com/avatar/1f400023980c374ae676?_k=1777273430885731487","ViCrit: A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs  \nXiyao Wang 1 ,2†, Zhengyuan Yang2†▽, Chao Feng3 ,4†  \nYuhang Zhou 1 , Xiaoyu Liu 1 , Yongyuan Liang 1 , Ming Li 1 , Ziyi Zang5 Chung-Ching Lin2 , Kevin Lin2 , Linjie Li2‡, Furong Huang 1‡, Lijuan Wang2‡  \n1University of Maryland, College Park 2Microsoft  \n3University of Michigan 4 Cornell University 5 Cardiff University  \n†First Authors ‡Equal Advising ▽ Project Lead  \n[xywang@umd.edu](xywang@umd.edu) [zhengyang@microsoft.com](zhengyang@microsoft.com)  \nAbstract  \nReinforcement learning (RL) has shown great effectiveness for fine-tuning large language models (LLMs) using tasks that are challenging yet easily verifiable, such as math reasoning or code generation. However, extending this success to visual perception in vision–language models (VLMs) has been impeded by the scarcity of vision-centric tasks that are simultaneously challenging and unambiguously verifiable. To this end, we introduce ViCrit (Visual Caption Hallucination Critic), an RL proxy task that trains VLMs to localize a subtle, synthetic visual hallucination injected into paragraphs of human-written image captions. Starting from a 200-word captions, we inject a single, subtle visual description error—altering a few words on objects, attributes, counts, or spatial relations—and task the model to pinpoint the corrupted span given the image and the modified caption. This formulation preserves the full perceptual difficulty while providing a binary, exactmatch reward that is easy to compute and unambiguous. Models trained with the ViCrit Task exhibit substantial gains across a variety of VL benchmarks. Crucially, the improvements transfer beyond natural-image training data to abstract image reasoning and visual math, showing promises of learning to perceive rather than barely memorizing seen objects. To facilitate evaluation, we further introduce ViCrit-Bench, a category-balanced diagnostic benchmark that systematically probes perception errors across diverse image domains and error types. Together, our results demonstrate that fine-grained hallucination criticism is an effective and generalizable objective for enhancing visual perception in VLMs.  \n1 Introduction  \nReinforcement learning (RL) has recently emerged as a dominate paradigm [17, 23] for fine-tuning large language models (LLMs) when training tasks are both challenging and automatically verifiable. Successful examples include mathematical reasoning tasks with concise numerical answers [19, 41], and software engineering problems [78, 38] whose correctness can be checked in a sandboxed environment. By focusing on tasks that strike this balance—sufficiently challenging to have room for improvements yet straightforward to grade deterministically—RL can explore the solution space effectively, extract genuinely useful strategies, and transfer those gains to broader domains.  \nDespite its success in textual reasoning, RL training with verifiable rewards has yet to demonstrate a comparable significance in improving the visual perception abilities of vision–language models (VLMs) . This is largely due to the lack of vision-centric tasks that are both perceptually  \n39th Conference on Neural Information Processing Systems (NeurIPS 2025) .  \nHuman-written caption:  \nThe image showcases a social gathering of Caucasian individuals, both male and female, ranging from middle age to about 60, seated at multiple tables inside a room that appears tobe a café or restaurant. The café’s walls are a light brown to mustard yellow, adorned with an eclectic mix of picture frames and flags, including one particularly striking black flag with curved white stitching that reads both \"true\" and \"false.\" There is a tall vertical window on the left side, offering a view of trees and parked cars outside.  \nHanging from the ceiling are two distinct light fixtures: a black wrought iron chandelier with six gold-colored bulbs, and a single glass pendant li","cbCaiqFcaKN0Gule","https://ap.wps.com/l/cbCaiqFcaKN0Gule","pdf",4324969,26,"English","# Abstract\n# 1 Introduction\n## Motivation for verifiable RL in VLM visual perception\n## ViCrit task design and reward formulation\n## ViCrit-Bench evaluation benchmark","[{\"question\":\"What problem does ViCrit address in visual perception for VLMs?\",\"answer\":\"ViCrit targets the lack of vision-centric RL tasks that are both perceptually challenging and unambiguously verifiable, which limits progress in improving VLM visual perception.\"},{\"question\":\"How does the ViCrit proxy task work?\",\"answer\":\"ViCrit starts from human-written image captions and injects a single subtle visual description error into the caption, then trains the model to locate the corrupted span given the image and modified caption.\"},{\"question\":\"What is ViCrit-Bench and why is it introduced?\",\"answer\":\"ViCrit-Bench is a category-balanced diagnostic benchmark that systematically probes perception errors across diverse image domains and different error types to facilitate evaluation.\"}]","ViCrit - A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs | PDF",66]