[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-140832-105":59,"doc-detail-140832-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","realrag-retrieval-augmented-realistic-image-generation-via-self-reflective-contrastive-learning","RealRAG - Retrieval-augmented Realistic Image Generation via Self-reflective Contrastive Learning","","Recent text-to-image generative models such as Stable Diffusion V3 and Flux achieve strong results, yet remain limited by the fixed knowledge encoded in their parameters trained on closed datasets. When prompts require fine-grained, unseen real-world objects, these models can produce hallucinations, distortions, and unrealistic details. RealRAG introduces a retrieval-augmented framework that learns to retrieve real-world images and compensates missing knowledge. A reflective retriever trained via self-reflective contrastive learning uses generator-aware negatives to improve realism and reduce distortion across diverse models, delivering significant performance gains including higher FID on benchmark datasets.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/realrag-retrieval-augmented-realistic-image-generation-via-self-reflective-contrastive-learning/140832/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/realrag-retrieval-augmented-realistic-image-generation-via-self-reflective-contrastive-learning/140832.png","ImageObject",300,407,{"name":92,"@type":93},"Arica Lee","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-13","2026-08-25",true,{"@type":102,"interactionType":103,"userInteractionCount":19},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does RealRAG address in text-to-image generation?","Question",{"text":112,"@type":113},"RealRAG targets hallucinations and distortions caused by generative models relying only on fixed parameters trained on closed datasets, which leads to poor handling of fine-grained and unseen real-world objects.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does RealRAG use retrieval to improve generated images?",{"text":117,"@type":113},"RealRAG retrieves real-world images to fill missing knowledge gaps and integrates fine-grained visual knowledge into the generation process, improving realism and reducing distortions.",{"name":119,"@type":110,"acceptedAnswer":120},"What is the role of the reflective retriever in RealRAG?",{"text":121,"@type":113},"The reflective retriever is trained with self-reflective contrastive learning to sample reflective negatives that are aligned with the generator’s stored visual memory, making retrieved augmentations compensate for missing knowledge more effectively.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},140832,1787644668,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":19,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":56,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":129,"read_time":143},8796096645457,"https://ap-avatar.wpscdn.com/avatar/800003749518d68ffe3?x-image-process=image/resize,m_fixed,w_180,h_180&k=1779345340919836971","RealRAG: Retrieval-augmented Realistic Image Generation via Self-reflective Contrastive Learning  \nYuanhuiyi Lyu 1 Xu Zheng 1 Lutao Jiang 1 Yibo Yan 1 Xin Zou 1 Huiyu Zhou 2 Linfeng Zhang 3 Xuming Hu 1 2 4  \narXiv :2502 .00848v3 [ cs .CV] 15 Sep 2025  \nAbstract  \nRecent text-to-image generative models, e.g., Stable Diffusion V3 and Flux, have achieved notable progress. However, these models are strongly restricted to their limited knowledge, a.k.a., their own fixed parameters, that are trained with closed datasets. This leads to significant hallucinations or distortions when facing fine-grained and unseen novel real-world objects, e.g., the appearance of the Tesla Cybertruck. To this end, we present the first real-object-based retrieval-augmented generation framework (RealRAG), which augments fine-grained and unseen novel object generation by learning and retrieving real-world images to overcome the knowledge gaps of generative models. Specifically, to integrate missing memory for unseen novel object generation, we train a reflective retriever by self-reflective contrastive learning, which injects the generator’s knowledge into the sef-reflective negatives, ensuring that the retrieved augmented images compensate for the model’s missing knowledge. Furthermore, the real-object-based framework integrates finegrained visual knowledge for the generative models, tackling the distortion problem and improving the realism for fine-grained object generation. Our Real-RAG is superior in its modular application to all types of state-of-the-art text-to-image generative models and also delivers remarkable performance boosts with all of them, such as again of 16.18% FID score with the auto-regressive model on the Stanford Car benchmark.  \n1The Hong Kong University of Science and Technology (Guangzhou) 2 Guangxi Key Laboratory of Digital Infrastructure, Guangxi Zhuang Autonomous Region Information Center 3 Shanghai Jiao Tong University 4The Hong Kong University of Science and Technology. Correspondence to: Xuming Hu \u003C[xuminghu@hkust-gz.edu.cn](xuminghu@hkust-gz.edu.cn) >.  \n(a)  \nPrompt:\"A Cybertruck is speeding along the Great Wall\"  \nText-to-Image Generative Model  \n(b)  \n(1)  \n(2)  \nSimilarity Score  \n Retrieve Prompt:  \n\"A Cybertruck is speeding along the Great Wall\"  \nRef. Image  \nText-to-Image Generative Model  \n\n| (1)\u003Cbr>(2) | | “A photo of cybertruck”\u003Cbr>“A truck is speeding along the Great Wall” |\n| --- | --- | --- |\n\n(c)  \nFigure 1 . (a) The pipeline of text-to-image generative models. (b) The framework of existing retrieval-augmented methods. (c) The framework of our proposed RealRAG.  \n1. Introduction  \nRecent text-to-image generators have achieved notable progress in image synthesis from the given textual prompts. There are three mainstream types of generative models, including the U-Net-based diffusion model (Rombach et al., 2022a ; Podell et al., 2023), the DiT-based diffusion model (Xiao et al., 2024 ; Sun et al., 2024), and the autoregressive model (Esser et al., 2024 ; BlackForest, 2024) . Typically, these models store all their visual memory (e.g., the appearance of Big Ben) implicitly in the parameters of the underlying neural network, requiring a lot of parameters(e.g., 10B) . Furthermore, similar to the hallucination problem of Large Language Models (LLMs) (OpenAI, 2023 ; Touvron et al., 2023), the large-scale text-to-image generative models also show the same problem. Some generated images include ghosting, distortions, and unnatural elements when generating specific real-world objects. Therefore, these problems motivate the development of textto-image generation models, which can integrate external visual knowledge (e.g., images from the web) to augment generative realism and accuracy for fine-grained and unseen novel object generation.  \nRetrieval-augmented generation (RAG) has shown promise in natural language processing (NLP) (Gao et al., 2023) . To enhance the specific knowledge and minimize the hallucination of LL","cbCaiv8pPorm7CCM","https://ap.wps.com/l/cbCaiv8pPorm7CCM","pdf",17529373,"English","# Abstract\n# Introduction\n## Motivation: limitations and hallucinations in generative models\n## Retrieval-augmented generation and its challenges in image synthesis\n# Proposed method: RealRAG and reflective retriever","[{\"question\":\"What problem does RealRAG address in text-to-image generation?\",\"answer\":\"RealRAG targets hallucinations and distortions caused by generative models relying only on fixed parameters trained on closed datasets, which leads to poor handling of fine-grained and unseen real-world objects.\"},{\"question\":\"How does RealRAG use retrieval to improve generated images?\",\"answer\":\"RealRAG retrieves real-world images to fill missing knowledge gaps and integrates fine-grained visual knowledge into the generation process, improving realism and reducing distortions.\"},{\"question\":\"What is the role of the reflective retriever in RealRAG?\",\"answer\":\"The reflective retriever is trained with self-reflective contrastive learning to sample reflective negatives that are aligned with the generator’s stored visual memory, making retrieved augmentations compensate for missing knowledge more effectively.\"}]","RealRAG - Retrieval-augmented Realistic Image Generation via Self-reflective Contrastive Learning | PDF",48]