[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-118244-en":3,"doc-seo-118244-105":30,"detail-sidebar-cat-0-en-105":91},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":4,"is_deleted":4,"is_public":20,"is_downloadable":20,"audit_status":20,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":14,"update_tm":28,"read_time":29},118244,1099514068035,"Ezra","https://ap-avatar.wpscdn.com/davatar_276721f389ce27ea32af1340a28f341c",8,"Research & Report","Towards Data-efficient Machine Learning Systems - Dissertation","Towards data-efficient machine learning systems investigates methods to reduce the amount of training data while maintaining or improving performance. It presents dataset pruning and sampling strategies for recommender systems, including SVP-CF and DATA-GENIE, to identify effective subsets for training and inference. The dissertation extends dataset pruning to large language model pretraining via ASK-LLM and DENSITY sampling, studying empirical gains, cost trade-offs, and model-size effects. It further explores data distillation for recommender systems using infinite-width autoencoders (∞-AE).","UC San Diego  \nUC San Diego Electronic Theses and Dissertations  \nTitle  \nTowards Data-efficient Machine Learning Systems  \nPermalink  \n[https://escholarship.org/uc/item/8nb955b8](https://escholarship.org/uc/item/8nb955b8)  \nAuthor  \nSachdeva, Noveen  \nPublication Date  \n2024  \nPeer reviewed|Thesis/dissertation  \n[eScholarship.org](eScholarship.org) Powered by the California Digital Library  \nUniversity of California  \nUNIVERSITY OF CALIFORNIA SAN DIEGO  \nTowards Data-efficient Machine Learning Systems  \nA dissertation submitted in partial satisfaction of the requirements for the degree Doctor of Philosophy  \nin  \nComputer Science  \nby  \nNoveen Sachdeva  \nCommittee in charge:  \nProfessor Julian McAuley, Chair  \nProfessor Zhiting Hu  \nProfessor Jingbo Shang  \nProfessor Lily Weng  \nCopyright  \nNoveen Sachdeva, 2024 All rights reserved.  \nThe Dissertation of Noveen Sachdeva is approved, and it is acceptable in quality and form for publication on microfilm and electronically.  \nUniversity of California San Diego  \n2024  \nDEDICATION  \nTo the hopefully unpredictable entropy in this world, universe, and beyond.  \nTABLE OF CONTENTS  \nDissertation Approval Page .................................................... iii  \nDedication .................................................................. iv  \nTable of Contents ............................................................ v  \nList of Figures ............................................................... viii  \nList of Tables ................................................................ xi  \nAcknowledgements ........................................................... xiii  \nVita ........................................................................ xv  \nAbstract of the Dissertation .................................................... xvi  \nIntroduction ................................................................. 1  \nChapter 1 Dataset Pruning for Recommender Systems .......................... 4  \n1.1 Introduction ......................................................... 5  \n1.2 Related Work ........................................................ 6  \n1.3 Sampling Collaborative Filtering Datasets ................................ 8  \n1.3.1 Problem Settings & Methods Compared ........................... 8  \n1.3.2 Sampling Strategies ............................................ 10  \n1.3.3 SVP-CF: Selection-Via-Proxy for CF data ......................... 12  \n1.3.4 Performance of a sampling strategy ............................... 16  \n1.3.5 Experiments .................................................. 16  \n1.4 DATA-GENIE: Which sampler is best for me? ............................. 21  \n1.4.1 Problem formulation ........................................... 21  \n1.4.2 Dataset representation .......................................... 22  \n1.4.3 Training & Inference ........................................... 23  \n1.4.4 Experiments .................................................. 26  \n1.5 Discussion .......................................................... 27  \nChapter 2 Dataset Pruning for Pretraining Large Language Models ................ 30  \n2.1 Introduction ......................................................... 31  \n2.1.1 Contributions ................................................. 33  \n2.2 Related Work ........................................................ 34  \n2.2.1 Coverage Sampling ............................................ 34  \n2.2.2 Quality-score Sampling ......................................... 35  \n2.3 Methods ............................................................ 36  \n2.3.1 ASK-LLM Sampling ........................................... 37  \n2.3.2 DENSITY Sampling ............................................ 39  \n2.3.3 Sampling Techniques ........................................... 43  \n2.3.4 Relationships Between Methods .................................. 43  \n2.4 Empirical Setup ...................................................... 45  \n2.4.1 Mo","cbCaicCzitNGuB6A","https://ap.wps.com/l/cbCaicCzitNGuB6A","pdf",3755574,1,172,"English","en",105,"# Introduction\n# Chapter 1 Dataset Pruning for Recommender Systems\n## Sampling Collaborative Filtering Datasets\n## DATA-GENIE: Which sampler is best for me?\n# Chapter 2 Dataset Pruning for Pretraining Large Language Models\n## ASK-LLM Sampling\n## DENSITY Sampling\n# Chapter 3 Data Distillation for Recommender Systems\n## ∞-AE: Infinite-width Autoencoders for Recommendation","[{\"question\":\"What core problem does the dissertation address?\",\"answer\":\"The dissertation focuses on improving data efficiency in machine learning systems by reducing the amount of training data needed while preserving performance.\"},{\"question\":\"Which methods are introduced for recommender systems?\",\"answer\":\"It proposes dataset pruning and sampling approaches, including SVP-CF and DATA-GENIE, to select effective samples for collaborative filtering training and inference.\"},{\"question\":\"How is data efficiency studied for large language model pretraining?\",\"answer\":\"The dissertation applies dataset pruning through sampling techniques such as ASK-LLM and DENSITY, then evaluates effects on reasoning, quality-score costs, and effective model size.\"}]","Towards Data-efficient Machine Learning Systems - Dissertation | PDF",1785682616,433,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":13,"keywords":34,"description":14,"schema_data":35,"social_meta":86,"head_meta":88,"extra_data":90,"updated_unix":28},"towards-data-efficient-machine-learning-systems-dissertation","",{"@graph":36,"@context":85},[37,54,68],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":20},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/document/","Document",2,{"item":49,"name":12,"@type":43,"position":50},"https://docshare.wps.com/document/research-report/",3,{"item":52,"name":13,"@type":43,"position":53},"https://docshare.wps.com/document/towards-data-efficient-machine-learning-systems-dissertation/118244/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":23,"description":14,"dateModified":62,"datePublished":62,"encodingFormat":61,"isAccessibleForFree":63,"interactionStatistic":64},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-08-02",true,{"@type":65,"interactionType":66,"userInteractionCount":4},"InteractionCounter",{"@type":67},"ViewAction",{"@type":69,"mainEntity":70},"FAQPage",[71,77,81],{"name":72,"@type":73,"acceptedAnswer":74},"What core problem does the dissertation address?","Question",{"text":75,"@type":76},"The dissertation focuses on improving data efficiency in machine learning systems by reducing the amount of training data needed while preserving performance.","Answer",{"name":78,"@type":73,"acceptedAnswer":79},"Which methods are introduced for recommender systems?",{"text":80,"@type":76},"It proposes dataset pruning and sampling approaches, including SVP-CF and DATA-GENIE, to select effective samples for collaborative filtering training and inference.",{"name":82,"@type":73,"acceptedAnswer":83},"How is data efficiency studied for large language model pretraining?",{"text":84,"@type":76},"The dissertation applies dataset pruning through sampling techniques such as ASK-LLM and DENSITY, then evaluates effects on reasoning, quality-score costs, and effective model size.","https://schema.org",{"og:url":52,"og:type":87,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":89,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":92},[93,97,101,105,110,115,120,123,128,131,135],{"id":20,"doc_module":4,"doc_module_name":46,"category_name":94,"show_sort_weight":95,"slug":96},"Story & Novel",90,"story-novel",{"id":47,"doc_module":4,"doc_module_name":46,"category_name":98,"show_sort_weight":99,"slug":100},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":46,"category_name":102,"show_sort_weight":103,"slug":104},"Exam",70,"exam",{"id":106,"doc_module":4,"doc_module_name":46,"category_name":107,"show_sort_weight":108,"slug":109},5,"Comic",60,"comic",{"id":111,"doc_module":4,"doc_module_name":46,"category_name":112,"show_sort_weight":113,"slug":114},6,"Technology",50,"technology",{"id":116,"doc_module":4,"doc_module_name":46,"category_name":117,"show_sort_weight":118,"slug":119},7,"Healthcare",40,"healthcare",{"id":11,"doc_module":4,"doc_module_name":46,"category_name":12,"show_sort_weight":121,"slug":122},30,"research-report",{"id":124,"doc_module":4,"doc_module_name":46,"category_name":125,"show_sort_weight":126,"slug":127},9,"Religion & Spirituality",20,"religion-spirituality",{"id":126,"doc_module":4,"doc_module_name":46,"category_name":129,"show_sort_weight":126,"slug":130},"World Cup","world-cup",{"id":132,"doc_module":4,"doc_module_name":46,"category_name":133,"show_sort_weight":132,"slug":134},10,"Lifestyle","lifestyle",{"id":136,"doc_module":4,"doc_module_name":46,"category_name":137,"show_sort_weight":106,"slug":138},19,"General","general"]