[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-148788-105":59,"doc-detail-148788-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","lever-lm-configuring-in-context-sequence-to-lever-large-vision-language-models","Lever LM - Configuring In-Context Sequence to Lever Large Vision Language Models","","Lever LM proposes a tiny language model (e.g., 67M parameters) to leverage much larger vision-language models (e.g., 9B parameters). It configures effective in-context demonstration sequences to strengthen in-context learning for LVLMs, viewing ICD sequence generation as a mirror of human sequential sentence composition and exploiting latent statistical patterns. A dataset of effective ICD sequences trains Lever-LM, and experiments improve performance on visual question answering and image captioning versus strong baselines.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/lever-lm-configuring-in-context-sequence-to-lever-large-vision-language-models/148788/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/lever-lm-configuring-in-context-sequence-to-lever-large-vision-language-models/148788.png","ImageObject",300,407,{"name":92,"@type":93},"Asher","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-17","2026-08-26",true,{"@type":102,"interactionType":103,"userInteractionCount":81},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does Lever LM address?","Question",{"text":112,"@type":113},"It addresses how to configure effective in-context demonstration (ICD) sequences to improve in-context learning performance of large vision-language models (LVLMs).","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How is ICD sequence configuration modeled in this work?",{"text":117,"@type":113},"It treats ICD sequence generation as a coherent, sequential process similar to human sentence composition, assuming effective sequences contain internal statistical patterns that can be learned.",{"name":119,"@type":110,"acceptedAnswer":120},"How are ICD sequences produced for new queries?",{"text":121,"@type":113},"After training, the learned Lever-LM generates new ICD sequences step by step for novel queries, enabling LVLMs to solve vision-language tasks via in-context learning.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},148788,1787786126,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":81,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},687197207639,"https://ap-avatar.wpscdn.com/davatar_a8503ba1806abce46bf441b54a3ca4cd","Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models  \nXu Yang 1 ,2 Yingzhe Peng 1 ,2 , Haoxuan Ma 1 ,2 , Shuo Xu 1 ,2 ,  \nChi Zhang3 , Yucheng Han4 , Hanwang Zhang4  \n1 Southeast University  \n2 Key Laboratory of New Generation Artificial Intelligence Technology &  \nIts Interdisciplinary Applications,(Southeast University),Ministry of Education  \n3 Westlake University  \n4 Nanyang Technological University  \n{xuyang_palm, yingzhe.peng, haoxuan-ma, [xushuo}@seu.edu.cn](xushuo}@seu.edu.cn)[ ](xushuo}@seu.edu.cn)[chizhang@westlake.edu.cn](chizhang@westlake.edu.cn) , [yucheng002@e.ntu.edu.sg](yucheng002@e.ntu.edu.sg) , [hanwangzhang@ntu.edu.sg](hanwangzhang@ntu.edu.sg)  \nAbstract  \nAs Archimedes famously said,“Give me a lever long enough and a fulcrum on which to place it, and I shall move the world”, in this study, we propose to use a tiny Language Model (LM), e.g., a Transformer with 67M parameters, to lever much larger Vision-Language Models (LVLMs) with 9B parameters. Specifically, we use this tiny Lever-LM to configure effective in-context demonstration (ICD) sequences to improve the In-Context Learinng (ICL) performance of LVLMs.  \nPrevious studies show that diverse ICD configurations like the selection and ordering of the demonstrations heavily affect the ICL performance, highlighting the significance of configuring effective ICD sequences. Motivated by this and by re-considering the the process of configuring ICD sequence, we find this is a mirror process of human sentence composition and further assume that effective ICD configurations may contain internal statistical patterns that can be captured by LeverLM. Then a dataset with effective ICD sequences is constructed to train Lever-LM.  \nAfter training, given novel queries, new ICD sequences are configured by the trained Lever-LM to solve vision-language tasks through ICL. Experiments show that these ICD sequences can improve the ICL performance of two LVLMs compared with some strong baselines in Visual Question Answering and Image Captioning, validating that Lever-LM can really capture the statistical patterns for levering LVLMs.  \nThe code is available at [https://github.com/ForJadeForest/Lever-LM](https://github.com/ForJadeForest/Lever-LM).  \n1 Introduction  \nWith the escalation in model size and training data [1–6], Large Language Models (LLMs) emerge the ability of In-Context Learning (ICL) [7–9] . ICL, akin to few-shot learning [10–12], utilizes a few exemplary In-Context Demonstrations (ICDs) to adapt LLMs to new tasks without gradient updates. This achievement in NLP has inspired researchers to similarly enhance Large Vision-Language Models (LVLMs) with ICL capabilities [13, 14] . However, just as in NLP, the effectiveness of ICL in LVLMs is significantly influenced by the configurations of ICDs, such as their selection and ordering [15–23] . Recent studies [24–26] have shown that this sensitivity in LVLMs is further exacerbated by the multimodal combinatorial complexity of vision and language data.  \nIn NLP, researchers employ various strategies to optimize in-context sequences to improve ICL performance, including retrieving representative examples as the ICDs [15, 27, 28] and re-ordering these ICDs based on specific principles [29, 21] . While these methods have shown improvements,  \n38th Conference on Neural Information Processing Systems (NeurIPS 2024) .  \nFigure 1: (a) The traditional ICD configuration methods separately select and order the ICDs, leading to sub-optimal ICL performance. (b) Our Lever-LM enables the step-by-step generation of ICD configurations and simultaneously considers the selection of ICDs and the ordering of ICD sequences.  \ntheir application remains largely confined to NLP and is less explored in the vision-language domain. Moreover, as shown in Fig. 1(a), the independent operations of retrieval and reordering often result in sub-optimal outcomes. A critical reconsideration of the ICD sequence generation reveal","cbCaio9OO3SEegqT","https://ap.wps.com/l/cbCaio9OO3SEegqT","pdf",2886804,28,"English","# Abstract\n# 1 Introduction\n## In-Context Learning and ICD configuration\n## Limitations of independent retrieval and reordering\n## Statistical-pattern motivation and Lever-LM idea","[{\"question\":\"What problem does Lever LM address?\",\"answer\":\"It addresses how to configure effective in-context demonstration (ICD) sequences to improve in-context learning performance of large vision-language models (LVLMs).\"},{\"question\":\"How is ICD sequence configuration modeled in this work?\",\"answer\":\"It treats ICD sequence generation as a coherent, sequential process similar to human sentence composition, assuming effective sequences contain internal statistical patterns that can be learned.\"},{\"question\":\"How are ICD sequences produced for new queries?\",\"answer\":\"After training, the learned Lever-LM generates new ICD sequences step by step for novel queries, enabling LVLMs to solve vision-language tasks via in-context learning.\"}]","Lever LM - Configuring In-Context Sequence to Lever Large Vision Language Models | PDF",71]