[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-327874-105":59,"doc-detail-327874-en":125},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":118,"head_meta":120,"extra_data":122,"updated_unix":124},105,"en","seeme-mitigating-hallucinations-in-large-vision-language-models-through-effective-visual-token-engineering-preprint","SeeMe - Mitigating Hallucinations in Large Vision-Language Models through Effective Visual Token Engineering - Preprint","","Large Vision-Language Models (LVLMs) have advanced visual understanding for captioning and visual question answering, yet they still produce hallucinations—outputs inconsistent with the provided image. Prior work mainly targets decoding, while ignoring visual-token sources that can mislead generation. SeeMe is a training-free framework that restructures visual tokens via a three-stage token engineering pipeline to suppress hallucination drivers while retaining informative evidence. Experiments on MME, POPE, and AMBER show consistent reductions in hallucinations and improved output consistency across four LVLMs.",{"@graph":69,"@context":117},[70,84,104],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/seeme-mitigating-hallucinations-in-large-vision-language-models-through-effective-visual-token-engineering-preprint/327874/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":98,"encodingFormat":97,"isAccessibleForFree":99,"interactionStatistic":100},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/seeme-mitigating-hallucinations-in-large-vision-language-models-through-effective-visual-token-engineering-preprint/327874.png","ImageObject",300,407,{"name":92,"@type":93},"Levi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-21",true,{"@type":101,"interactionType":102,"userInteractionCount":8},"InteractionCounter",{"@type":103},"ViewAction",{"@type":105,"mainEntity":106},"FAQPage",[107,113],{"name":108,"@type":109,"acceptedAnswer":110},"What stages are included in SeeMe’s token engineering pipeline?","Question",{"text":111,"@type":112},"The pipeline includes Selection (pruning via cross-modal attention), Merging (similarity-driven fusion to form a high-quality token pool), and a final Selection (choosing tokens best aligned with language context).","Answer",{"name":114,"@type":109,"acceptedAnswer":115},"Which benchmarks and models are used to evaluate SeeMe?",{"text":116,"@type":112},"SeeMe is evaluated on MME, POPE, and AMBER benchmarks across four LVLMs, including LLaVA-1.5, LLaVA-NEXT, INF-MLLM, and mPLUG-Owl2.","https://schema.org",{"og:url":83,"og:type":119,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":121,"canonical":83},"index,follow",{"doc_id":123,"site_id":62},327874,1790002170,{"code":4,"msg":5,"data":126},{"doc_id":123,"user_id":127,"nickname":92,"user_avatar":128,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":129,"file_id":130,"file_url":131,"file_type":132,"file_size":133,"view_count":8,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":134,"language":135,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":67,"update_tm":139,"read_time":41},7971461740909,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","arXiv :2607 .04 163v 1 [ cs .CV] 5 Jul 2026  \nSeeMe: Mitigating Hallucinations in Large Vision-Language Models through Effective Visual Token Engineering  \nKai Tang 1,4,* Jinhao You2,*  \n1  \nBohua Zhang 1,3 Yichen Guo 1,4 Yiding Sun 1 Dongxu Zhang5 Xiande Huang7,† Shanghang Zhang 1,†  \nState Key Laboratory of Multimedia Information Processing,  \nSchool of Computer Science, Peking University  \n2 University of Pennsylvania  \n3 University of Electronic Science and Technology of China  \n4 Nanyang Technological University  \n5 Tsinghua University  \n6 The Chinese University of Hong Kong, Shenzhen  \n7 De Artificial Intelligence Lab  \n*Equal contribution. †Corresponding authors.  \nChenxi Li6  \nAbstract  \nLarge Vision-Language Models (LVLMs) have achieved remarkable progress in visual understanding tasks such as image captioning and visual question answering. However, they remain susceptible to hallucinations, generating content that is inconsistent with the actual visual input.  \nExisting methods primarily intervene at the decoding stage, while overlooking a critical source of hallucinations: irrelevant or noisy visual tokens that mislead the decoding process. To address this issue, we propose SeeMe, a training-free framework that introduces the concept of feature engineering from traditional machine learning into LVLMs. SeeMe restructures visual tokens through a three-stage token engineering process to suppress hallucination sources while preserving informative visual evidence. Experiments on MME, POPE, and AMBER benchmarks across four LVLMs demonstrate that SeeMe consistently reduces hallucinations and improves output consistency, providing a novel perspective for mitigating hallucinations in LVLMs.  \n1. Introduction  \nLarge Vision-Language Models (LVLMs) have achieved remarkable success in open-ended visual understanding tasks, including image captioning, visual question answering, and multimodal dialogue (Liu et al., 2023b ; Hu et al., 2023 ; Zhu & et al., 2023 ; Bai et al., 2023 ; Ye et al., 2024) . De  \nContact: Kai Tang \u003C[kaitang030113@gmail.com](kaitang030113@gmail.com) >; Correspondence: Shanghang Zhang \u003C[shanghang@pku.edu.cn](shanghang@pku.edu.cn) >. Preprint. July 7, 2026.  \nFigure 1. Comparison of visual token handling strategies. Token engineering reconstructs informative representations, leading to more accurate answers.  \nspite these advances, LVLMs remain prone to hallucination—generating content that is inconsistent with the actual visual input (Liu et al., 2024b) . This phenomenon severely undermines their reliability in practical applications, particularly in safety-critical or factual scenarios (Hartsock & Rasool, 2024 ; Zhou et al., 2024 ; Ma et al., 2024) .  \nRecent studies have attributed hallucinations in LVLMs to several factors, including over-reliance on statistical biases in training data (Zhou et al., 2023b ; Chen et al., 2024b), the dominance of language priors over weak visual grounding (Han et al., 2022 ; Guan et al., 2024), and the dilution of cross-modal attention in deeper layers (An et al., 2025) . To mitigate these issues, recent work has explored training-free strategies that intervene during inference. These methods typically operate on the language decoder’s internal dynamics: DoLa (Chuang et al., 2023) compares early and late layer outputs, VCD (Leng et al., 2024) contrasts original and distorted visual inputs, DAMO (Wang et al., 2025) accumulates past activations to stabilize hidden states, and DCLA (Tang et al., 2026) enforces inter-layer consistency  \nto reduce semantic drift.  \nAlthough these methods have shown effectiveness, they primarily operate on the language decoder’s internal dynamics, overlooking a key factor: visual tokens themselves are often a primary source of hallucinations. In many cases, hallucinations arise from irrelevant or noisy visual tokens passed from the visual encoder to the language decoder, leading the model to generate outputs inconsistent with the actual visual co","cbCaiqJ6ADf0GquJ","https://ap.wps.com/l/cbCaiqJ6ADf0GquJ","pdf",2225309,12,"English","# Introduction\n## Hallucination sources in LVLMs\n## Training-free decoding interventions\n## Visual-token manipulation approaches\n## Proposed token engineering framework (SeeMe)","[{\"question\":\"What stages are included in SeeMe’s token engineering pipeline?\",\"answer\":\"The pipeline includes Selection (pruning via cross-modal attention), Merging (similarity-driven fusion to form a high-quality token pool), and a final Selection (choosing tokens best aligned with language context).\"},{\"question\":\"Which benchmarks and models are used to evaluate SeeMe?\",\"answer\":\"SeeMe is evaluated on MME, POPE, and AMBER benchmarks across four LVLMs, including LLaVA-1.5, LLaVA-NEXT, INF-MLLM, and mPLUG-Owl2.\"}]","SeeMe - Mitigating Hallucinations in Large Vision-Language Models through Effective Visual Token Engineering - Preprint | PDF",1789978237]