[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-327881-105":3,"detail-sidebar-cat-0-en-105":79,"doc-detail-327881-en":129},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":72,"head_meta":74,"extra_data":76,"updated_unix":78},105,"en","detecting-hallucinations-in-retrieval-augmented-generation-through-grounding-aware-sensitivity-by-perturbation-gasp","Detecting Hallucinations in Retrieval-Augmented Generation through Grounding-Aware Sensitivity by Perturbation (GASP)","","Retrieval-augmented generation (RAG) reduces hallucinations but detectors often output a single score without showing which answer sentence is unsupported or why. Grounding-Aware Sensitivity by Perturbation (GASP) is a span-level detector that fixes the answer and re-scores it under full context, no context, and with each retrieved chunk removed, measuring log-likelihood drops and Jensen-Shannon divergences as grounding sensitivity. GASP finds grounded sentences collapse when support is removed while hallucinated ones remain stable, evaluated on RAGTruth, TofuEval, and RAGBench with leakage-clean protocols; it substantially outperforms baselines and transfers selectively.",{"@graph":14,"@context":71},[15,34,54],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/research-report/","Research & Report",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/detecting-hallucinations-in-retrieval-augmented-generation-through-grounding-aware-sensitivity-by-perturbation-gasp/327881/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":48,"encodingFormat":47,"isAccessibleForFree":49,"interactionStatistic":50},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/detecting-hallucinations-in-retrieval-augmented-generation-through-grounding-aware-sensitivity-by-perturbation-gasp/327881.png","ImageObject",300,407,{"name":42,"@type":43},"Levi","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-21",true,{"@type":51,"interactionType":52,"userInteractionCount":4},"InteractionCounter",{"@type":53},"ViewAction",{"@type":55,"mainEntity":56},"FAQPage",[57,63,67],{"name":58,"@type":59,"acceptedAnswer":60},"What problem does GASP address in retrieval-augmented generation (RAG)?","Question",{"text":61,"@type":62},"GASP targets detectors that give only a single answer-level score, without indicating which specific sentence is unsupported or providing an explanation.","Answer",{"name":64,"@type":59,"acceptedAnswer":65},"How does GASP measure whether an answer sentence is grounded?",{"text":66,"@type":62},"GASP keeps the answer fixed and re-scores it under full context, no context, and with each retrieved chunk removed, using log-likelihood drops and Jensen-Shannon divergences to quantify grounding sensitivity.",{"name":68,"@type":59,"acceptedAnswer":69},"What distinguishes grounded from hallucinated sentences under perturbation?",{"text":70,"@type":62},"The likelihood of a grounded sentence collapses when its supporting passage is removed, while a hallucinated sentence is almost unaffected by the same removal.","https://schema.org",{"og:url":32,"og:type":73,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":75,"canonical":32},"index,follow",{"doc_id":77,"site_id":7},327881,1789978256,{"code":4,"msg":80,"data":81},"success",[82,86,90,94,99,104,109,113,118,121,125],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":83,"show_sort_weight":84,"slug":85},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":87,"show_sort_weight":88,"slug":89},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":91,"show_sort_weight":92,"slug":93},"Exam",70,"exam",{"id":95,"doc_module":4,"doc_module_name":25,"category_name":96,"show_sort_weight":97,"slug":98},5,"Comic",60,"comic",{"id":100,"doc_module":4,"doc_module_name":25,"category_name":101,"show_sort_weight":102,"slug":103},6,"Technology",50,"technology",{"id":105,"doc_module":4,"doc_module_name":25,"category_name":106,"show_sort_weight":107,"slug":108},7,"Healthcare",40,"healthcare",{"id":110,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":111,"slug":112},8,30,"research-report",{"id":114,"doc_module":4,"doc_module_name":25,"category_name":115,"show_sort_weight":116,"slug":117},9,"Religion & Spirituality",20,"religion-spirituality",{"id":116,"doc_module":4,"doc_module_name":25,"category_name":119,"show_sort_weight":116,"slug":120},"World Cup","world-cup",{"id":122,"doc_module":4,"doc_module_name":25,"category_name":123,"show_sort_weight":122,"slug":124},10,"Lifestyle","lifestyle",{"id":126,"doc_module":4,"doc_module_name":25,"category_name":127,"show_sort_weight":95,"slug":128},19,"General","general",{"code":4,"msg":80,"data":130},{"doc_id":77,"user_id":131,"nickname":42,"user_avatar":132,"doc_module":4,"category_id":110,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":133,"file_id":134,"file_url":135,"file_type":136,"file_size":137,"view_count":4,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":138,"language":139,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":12,"update_tm":78,"read_time":143},7971461740909,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","arXiv :2607 .04223v 1 [ cs .CL] 5 Jul 2026  \nDetecting Hallucinations in Retrieval-Augmented Generation through Grounding-Aware Sensitivity by Perturbation (GASP)  \nMohamed Aly Bouke \\# 1,*  \n1Centre for Intelligent Cloud Computing, CoE for Advanced Cloud, Faculty of Information Science and Technology,  \nMultimedia University, Jalan Ayer Keroh Lama, Bukit Beruang, 75450, Melaka, Malaysia  \n*[bouke@ieee.org](bouke@ieee.org) , [alybouke@mmu.edu.my](alybouke@mmu.edu.my)  \nResearch Article, July 7, 2026  \nAbstract  \nRetrieval-augmented generation (RAG) reduces but does not eliminate hallucination, and existing detectors return a single answer-level score that does not indicate which sentence is unsupported, or why. To close this gap, we introduce Grounding-Aware Sensitivity by Perturbation (GASP), a span-level detector that scores each answer sentence by how strongly its likelihood depends on the retrieved evidence, a quantity we term grounding sensitivity. GASP holds the answer fixed and re-scores it under the full context, under no context, and with each chunk removed, then measures the log-likelihood drops and Jensen-Shannon divergences (JSD) . The likelihood of a grounded sentence collapses once its supporting passage is removed, whereas a hallucinated sentence is almost unaffected, a contrast we interpret by casting decoding as a random nonlinear iterated function system (RNIFS) . We evaluate GASP on three benchmarks (RAGTruth, TofuEval, RAGBench) with three instruction-tuned scorers from two model families (Qwen2.5-0.5B, Qwen2.5-1.5B, and SmolLM2-1.7B) under a leakage-clean protocol. On RAGTruth it reaches a response-level area under the receiver operating characteristic (ROC) curve (AUC) of about 0.73 and a span-level AUC of about 0.67, improving significantly over perplexity and by clear margins over length, whole-context natural language inference (NLI), and self-consistency baselines. The only baseline competitive at the span level is a well-configured chunk-level entailment verifier, which requires a separate model, whereas a training-free threshold on the grounding features matches the trained classifier without labeled data and serves as the default detector. Beyond RAGTruth, the signal transfers to TofuEval but not to short-answer question answering in RAGBench, showing GASP is best suited to outputs constructed from the retrieved context rather than answers recoverable from parametric knowledge.  \nKeywords: hallucination detection, retrieval-augmented generation, context perturbation, explainable AI, large language models  \n1 Introduction   \nLarge language models (LLMs) are increasingly deployed with RAG, where a retriever supplies passages that the model conditions on when answering a query [1] . Grounding generation in retrieved evidence narrows the gap between what a model asserts and what can be verified, and it has become the dominant pattern for question answering, summarization, and assistant systems in knowledge-intensive settings. Retrieval injects upto-date and domain-specific information that a fixed parametric model cannot hold, and it offers a route to attribution, since each claim can in principle be traced to a retrieved passage. The pattern is now spreading into settings where the cost of an error is high, such as clinical decision support, legal research, and financial analysis, which raises the stakes of any unsupported statement the system produces and calls for checking grounding at the level of the individual claim.  \nYet retrieval does not remove the core failure of generative models. They continue to produce fluent statements that the supplied evidence does not support, a phenomenon usually called hallucination [2], [3] . The model remains free to blend its parametric prior with the retrieved text, to over-generalize from a partial match, or to contradict a passage it has nominally read. A single unsupported sentence that reads as confidently asa true one can cause real harm. Consider a clinic","cbCaiduGOWBcnsPg","https://ap.wps.com/l/cbCaiduGOWBcnsPg","pdf",1274481,23,"English","# Introduction\n## Motivation for hallucination detection in RAG\n## Limitations of existing detectors\n## Central observation and approach\n## Evaluation setup and results","[{\"question\":\"What problem does GASP address in retrieval-augmented generation (RAG)?\",\"answer\":\"GASP targets detectors that give only a single answer-level score, without indicating which specific sentence is unsupported or providing an explanation.\"},{\"question\":\"How does GASP measure whether an answer sentence is grounded?\",\"answer\":\"GASP keeps the answer fixed and re-scores it under full context, no context, and with each retrieved chunk removed, using log-likelihood drops and Jensen-Shannon divergences to quantify grounding sensitivity.\"},{\"question\":\"What distinguishes grounded from hallucinated sentences under perturbation?\",\"answer\":\"The likelihood of a grounded sentence collapses when its supporting passage is removed, while a hallucinated sentence is almost unaffected by the same removal.\"}]","Detecting Hallucinations in Retrieval-Augmented Generation through Grounding-Aware Sensitivity by Perturbation (GASP) | PDF",58]