[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-462746-105":3,"detail-sidebar-cat-0-en-105":81,"doc-detail-462746-en":130},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":74,"head_meta":76,"extra_data":78,"updated_unix":80},105,"en","machine-learningdriven-language-assessment-method-for-rapidly-creating-valid-reliable-and-secure-proficiency-tests","Machine Learning–Driven Language Assessment - Method for Rapidly Creating Valid, Reliable, and Secure Proficiency Tests","","We describe a method for rapidly creating language proficiency assessments, with experimental evidence that resulting tests are valid, reliable, and secure. The approach is the first to use machine learning and natural language processing to induce proficiency scales from a given standard, then use linguistic models to estimate item difficulty for computer-adaptive testing. This reduces expensive human pilot testing and enables a large item bank, demonstrated with the Duolingo English Test.",{"@graph":14,"@context":73},[15,34,56],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/research-report/","Research & Report",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/machine-learningdriven-language-assessment-method-for-rapidly-creating-valid-reliable-and-secure-proficiency-tests/462746/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/machine-learningdriven-language-assessment-method-for-rapidly-creating-valid-reliable-and-secure-proficiency-tests/462746.png","ImageObject",300,407,{"name":42,"@type":43},"kopisore","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-10-09","2026-09-30",true,{"@type":52,"interactionType":53,"userInteractionCount":55},"InteractionCounter",{"@type":54},"ViewAction",6,{"@type":57,"mainEntity":58},"FAQPage",[59,65,69],{"name":60,"@type":61,"acceptedAnswer":62},"How does the method reduce the need for expensive human pilot testing?","Question",{"text":63,"@type":64},"It automatically induces proficiency scales using machine learning and natural language processing and uses linguistic models to estimate item difficulty for computer-adaptive testing, reducing reliance on human-graded pilot studies.","Answer",{"name":66,"@type":61,"acceptedAnswer":67},"What psychometric framework and model are used for item scoring?",{"text":68,"@type":64},"The work uses a logistic item response function, specifically the Rasch model, to express the probability of a correct response as a function of item difficulty and examinee ability.",{"name":70,"@type":61,"acceptedAnswer":71},"What exam is developed using these methods?",{"text":72,"@type":64},"The approach is used to develop the Duolingo English Test, and the paper reports that its scores align significantly with other high-stakes English assessments while remaining reliable and secure.","https://schema.org",{"og:url":32,"og:type":75,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":77,"canonical":32},"index,follow",{"doc_id":79,"site_id":7},462746,1791330696,{"code":4,"msg":82,"data":83},"success",[84,88,92,96,101,105,110,114,119,122,126],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":85,"show_sort_weight":86,"slug":87},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":89,"show_sort_weight":90,"slug":91},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":93,"show_sort_weight":94,"slug":95},"Exam",70,"exam",{"id":97,"doc_module":4,"doc_module_name":25,"category_name":98,"show_sort_weight":99,"slug":100},5,"Comic",60,"comic",{"id":55,"doc_module":4,"doc_module_name":25,"category_name":102,"show_sort_weight":103,"slug":104},"Technology",50,"technology",{"id":106,"doc_module":4,"doc_module_name":25,"category_name":107,"show_sort_weight":108,"slug":109},7,"Healthcare",40,"healthcare",{"id":111,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":112,"slug":113},8,30,"research-report",{"id":115,"doc_module":4,"doc_module_name":25,"category_name":116,"show_sort_weight":117,"slug":118},9,"Religion & Spirituality",20,"religion-spirituality",{"id":117,"doc_module":4,"doc_module_name":25,"category_name":120,"show_sort_weight":117,"slug":121},"World Cup","world-cup",{"id":123,"doc_module":4,"doc_module_name":25,"category_name":124,"show_sort_weight":123,"slug":125},10,"Lifestyle","lifestyle",{"id":127,"doc_module":4,"doc_module_name":25,"category_name":128,"show_sort_weight":97,"slug":129},19,"General","general",{"code":4,"msg":82,"data":131},{"doc_id":79,"user_id":132,"nickname":42,"user_avatar":133,"doc_module":4,"category_id":111,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":55,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":139,"language":140,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":12,"update_tm":144,"read_time":145},962090880963,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","Machine Learning–Driven Language Assessment  \nBurr Settlesand Geoffrey T. LaFlair Duolingo Pittsburgh, PA USA {burr,[geoff](geoff}@duolingo.com)[}](geoff}@duolingo.com)[@duolingo.com](geoff}@duolingo.com)  \nMasato Hagiwara∗  \nOctanove Labs Seattle, WA USA [masato@octanove.com](masato@octanove.com)  \nAbstract  \nWe describe a method for rapidly creating language proficiency assessments, and provide experimental evidence that such tests can be valid, reliable, and secure. Our approach is the first to use machine learning and natural language processing to induce proficiency scales basedona given standard, and then use linguistic models to estimate item difficulty directly for computer-adaptive testing. This alleviatesthe need for expensive pilot testing with human subjects. We used these methods to develop an online proficiency exam called the Duolingo English Test, and demonstrate that its scores align significantly with otherhigh-stakes English assessments. Furthermore, our approach produces test scores that are highly reliable, while generating item banks large enough to satisfy security requirements.  \n1 Introduction  \nLanguage proficiency testing is an increasingly important part of global society. The need to demonstrate language skills—often through standardized testing—is now required in many situations for access to higher education, immigration, and employment opportunities. However, standardized tests are cumbersome to create and maintain. Lane et al. (2016) and the Standards for Educational and Psychological Testing (AERA et al., 2014) describe many of the proceduresand requirements for planning, creating, revising, administering, analyzing, and reporting on high-stakes tests and their development.  \nIn practice, test items are often first written by subject matter experts, and then ‘‘pilot tested’’with a large number of human subjects for psy-  \n∗ Research conducted at Duolingo.  \nchometric analysis. This labor-intensive process often restricts the number ofitems that can feasibly be created, which in turn poses a threat to security: Items may be copied and leaked, or simply used too often (Cau, 2015; Dudley et al., 2016) . Security can be enhanced through computeradaptive testing (CAT), by which a subset of items are administered in a personalized way (based on examinees’ performance on previous items). Because the item sequences are essentially unique for each session, there is no single test form to obtain and circulate (Wainer, 2000), but these security benefits only hold if the item bank is large enough to reduce item exposure (Way, 1998) . This further increases the burden on item writers, and also requires significantly more item pilot testing.  \nFor the case of language assessment, we tackle both of these development bottlenecks using machine learning (ML) and natural language processing (NLP) . In particular, we propose the use of test item formats that can be automatically created, graded, and psychometrically analyzed using ML/NLP techniques. This solves the ‘‘cold start’’ problem in language test development, by relaxing manual item creation requirements and alleviating the need for human pilot testing altogether.  \nIn the pages that follow, we first summarize the important concepts from language testing and psychometrics (§2), and then describe our ML/NLP methods to learn proficiency scales for both words (§3) and long-form passages (§4) . We then present evidence for the validity, reliability, and security of our approach using results from the Duolingo English Test, an online, operational English proficiency assessment developed using these methods (§5) . After summarizing other related work (§6), we conclude with a discussion of limitations and future directions (§7) .  \n247  \nTransactions of the Association for Computational Linguistics, vol. 8, pp. 247–263, 2020. [https://doi.org/10.1162/tacl](https://doi.org/10.1162/tacl a 00310)[ a ](https://doi.org/10.1162/tacl a 00310)[00310](https://doi.org/10.","cbCaijo9bmrxel3V","https://ap.wps.com/l/cbCaijo9bmrxel3V","pdf",890091,17,"English","# Abstract\n# 1 Introduction\n# 2 Background\n## 2.1 Item Response Theory (IRT)","[{\"question\":\"How does the method reduce the need for expensive human pilot testing?\",\"answer\":\"It automatically induces proficiency scales using machine learning and natural language processing and uses linguistic models to estimate item difficulty for computer-adaptive testing, reducing reliance on human-graded pilot studies.\"},{\"question\":\"What psychometric framework and model are used for item scoring?\",\"answer\":\"The work uses a logistic item response function, specifically the Rasch model, to express the probability of a correct response as a function of item difficulty and examinee ability.\"},{\"question\":\"What exam is developed using these methods?\",\"answer\":\"The approach is used to develop the Duolingo English Test, and the paper reports that its scores align significantly with other high-stakes English assessments while remaining reliable and secure.\"}]","Machine Learning–Driven Language Assessment - Method for Rapidly Creating Valid, Reliable, and Secure Proficiency Tests | PDF",1790765083,43]