[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-128865-105":59,"doc-detail-128865-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","training-efficient-agents-for-long-term-decision-making","Training Efficient Agents for Long-Term Decision Making","","Reinforcement learning has progressed from simple simulators to real robots and open-world games, yet agents still suffer from prohibitively low sample efficiency, underuse priors encoded in foundation models, and rapidly forget most observations after a small number of steps. This thesis advances a unifying goal: efficiently train long-horizon decision-making agents through three successive contributions. It improves sample efficiency by prioritizing the most informative transitions, enabling offline RL to learn safe, high-performing policies.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/training-efficient-agents-for-long-term-decision-making/128865/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/training-efficient-agents-for-long-term-decision-making/128865.png","ImageObject",300,407,{"name":92,"@type":93},"Aria","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-18","2026-08-06",true,{"@type":102,"interactionType":103,"userInteractionCount":34},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does the thesis address in reinforcement learning agents?","Question",{"text":112,"@type":113},"Agents learn with very low sample efficiency, fail to effectively leverage priors from foundation models, and tend to forget much of what they experience after only a few hundred steps.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"What is the main unifying goal of the thesis?",{"text":117,"@type":113},"The thesis aims to efficiently train long-term decision-making agents via three successive contributions.",{"name":119,"@type":110,"acceptedAnswer":120},"How does the thesis improve sample efficiency in Chapter 3?",{"text":121,"@type":113},"It improves sample efficiency by re-weighting experience toward the transitions that are most informative, using an ensemble-based uncertainty criterion to selectively upsample rare interactions that reveal causal structure.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},128865,1786004026,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":34,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},2336474459895,"https://ap-avatar.wpscdn.com/avatar/22000baeef7a5ed0655?x-image-process=image/resize,m_fixed,w_180,h_180&k=1786071322749376916","Training Efficient Agents for Long-Term Decision  \nMaking  \nGunshi Gupta Supervisor: Prof. Yarin Gal  \nLady Margaret Hall University of Oxford  \nJuly 2025  \nAcknowledgements  \nI would like to express my sincere gratitude to everyone who has supported me throughout my D.Phil. journey at Oxford.  \nFirst and foremost, I am deeply thankful to my advisor, Yarin Gal, for fostering an open, collaborative, and exploratory environment at OATML. Yarin’s guidance and trust gave me the freedom to pursue my research interests while learning from the incredible community he helped build. I especially appreciated his balance of thoughtful feedback and allowing me the space to develop my own ideas, as well as the lab culture he shaped — one that encouraged intellectual curiosity, collaboration, and exploration. I am grateful to Tim G. Rudner for his excellent mentorship and advice over the years—supervising two of my projects, Tim taught me invaluable lessons about positioning and presenting research with clarity and polish. I am fortunate to have been advised and mentored by Adrien Gaidonand Rowan McAllister from TRI on my first research project. Our discussions on robotics and related challenges were formative, and I appreciated their guidance and support throughout. I would also like to thank Rahaf Aljundi from Toyota Motors Europe for her mentorship, openness, encouragement, and collaborative spirit, which made pursuing several interesting and diverse research directions possible. I always enjoyed our brainstorming sessions—her thoughtful questions and openness to exploring different ideas made them both productive and fun.  \nA special thanks to my main collaborator and partner throughout this journey, my husband Karmesh Yadav—none of this would have been possible without his insight, encouragement, and shared problem-solving, and he made the doctoral journey twice as fun as it might have been if we hadn’t embarked on it together.  \nTo all my OATML labmates—Lisa, Shreshth, Kelsey, Jannik, Angus, Andrew, Mo, Joost, Tim, Milad, Angelos, Clare, Tim, Panos, Pascal, Hazel, Daniella, Kunal, Yonatan, Matt, Luke and others—thank you for all the thoughtful discussions and for being a supportive research community. Lisa, in particular, for the long coffee  \nUniversity of Oxford  \nwalks and conversations, for the many small things I learnt from her, and for being such a grounding presence throughout my time at OATML—I’m glad we’ll be in the same city again soon.  \nI would like to thank Katrina for managing the many moving parts of OATML so smoothly and generously, all while pursuing her own studies at Oxford. Her support and involvement greatly improved our lab’s functioning. And to Xiaowen Dong, for looking out for students at LMH and regularly checking in—thank you. I am grateful to my transfer and confirmation assessors, Joao Henriques and Jan-Peter Calliess, for their feedback during key milestones.  \nMy internship at Microsoft Research was another important part of this journey. Iam thankful to Katja Hofmann, Sam Devlin, Tim Pearce, Tabish Rashid, Anssi Kanervisto, Tarun Gupta, Lukas, Raluca, Udit, and the rest of the Gaming Intelligence team for the opportunity to learn and contribute during my time there. I also valued working with and learning from Cong Lu on different reinforcement learning projects; his positivity and motivation made every interaction rewarding. My thanks as well to Yusuf, with whom I collaborated on my latest project, and to Zsolt Kira and Dhruv Batra for their informal advising and perspectives.  \nTo the AIMS CDT cohort—Kelsey, Shreshth, Mathew, Patrick, Yash, Benedetta, Seb, Aleks, and others—I appreciated sharing this experience with all of you. And to Wendy Poole, for being the ever-reliable CDT course administrator who helped navigate everything behind the scenes with kindness and humour.  \nOn a personal note, I am endlessly grateful to my parents Anju and Rakesh, and my sister, Sachi, for their unwavering support an","cbCaib5tJWp9WoI0","https://ap.wps.com/l/cbCaib5tJWp9WoI0","pdf",37494404,203,"English","# Abstract\n## Chapter 3\n## Unifying agenda: efficient long-term decision-making agents\n## Efficiently improving sample efficiency through transition re-weighting","[{\"question\":\"What problem does the thesis address in reinforcement learning agents?\",\"answer\":\"Agents learn with very low sample efficiency, fail to effectively leverage priors from foundation models, and tend to forget much of what they experience after only a few hundred steps.\"},{\"question\":\"What is the main unifying goal of the thesis?\",\"answer\":\"The thesis aims to efficiently train long-term decision-making agents via three successive contributions.\"},{\"question\":\"How does the thesis improve sample efficiency in Chapter 3?\",\"answer\":\"It improves sample efficiency by re-weighting experience toward the transitions that are most informative, using an ensemble-based uncertainty criterion to selectively upsample rare interactions that reveal causal structure.\"}]","Training Efficient Agents for Long-Term Decision Making | PDF",512]