[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-474489-105":59,"doc-detail-474489-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","rater-reliability-and-rating-scale-utility-for-the-ap-japanese-computer-simulated-conversation-task-evaluation-inference","Rater Reliability and Rating Scale Utility for the AP Japanese Computer-Simulated Conversation Task - Evaluation Inference","","This study examined the validity of scoring procedures for the AP Japanese computer-simulated conversation task through an argument-based approach, emphasizing rater reliability and rating scale functioning. Data came from 102 high school students using a test simulation, where three raters scored performances on a shared 7-point scale. Partial Credit Rasch model analyses assessed scores across raters and speech acts. Findings supported rater reliability, with limited evidence for intended rating scale functioning.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/rater-reliability-and-rating-scale-utility-for-the-ap-japanese-computer-simulated-conversation-task-evaluation-inference/474489/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/rater-reliability-and-rating-scale-utility-for-the-ap-japanese-computer-simulated-conversation-task-evaluation-inference/474489.png","ImageObject",300,407,{"name":92,"@type":93},"WPS_1790064749","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-07","2026-09-30",true,{"@type":102,"interactionType":103,"userInteractionCount":39},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What aspects of scoring validity were investigated for the AP Japanese conversation task?","Question",{"text":112,"@type":113},"The study focused on rater reliability and the functioning of the rating scale, using an argument-based approach to validity.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How was the task administered and how were performances scored?",{"text":117,"@type":113},"A computer simulation collected responses from 102 high school students, and three raters scored performances using a common 7-point holistic scale.",{"name":119,"@type":110,"acceptedAnswer":120},"What were the main results regarding rater reliability and rating scale utility?",{"text":121,"@type":113},"Results supported rater reliability, but they provided only limited support for how well the rating scale functioned as intended.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},474489,1790870156,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":39,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":145},3985747859154,"https://ap-avatar.wpscdn.com/davatar_276721f389ce27ea32af1340a28f341c","Rater Reliability and Rating Scale Utility for the AP Japanese Computer-Simulated Conversation Task:  \nEvaluation Inference  \nNana Suzumura-Smith  \nCalifornia State University, Long Beach  \nAbstract  \nThis study examined the validity of the scoring procedures for the AP Japanese conversation task using an argument-based approach, with a focus on rater reliability and rating scale functioning. Data were collected from 102 highschool students through a test simulation, with three raters scoring the performances using a common 7-point scale. Test scores were analyzed across raters and speech acts using the Partial Credit Rasch model. Results provided support for rater reliability but only limited support for the intended functioning of the rating scale. To enhance task validity, three potential modifications were proposed: controlling speech act types and numbers, reducing the number of score categories, and modifying the scoring procedure. This study sheds light on the validity argument for the AP Japanese conversation  \nJNCOLCTL VOL 38  \nRater Reliability and Rating Scale Utility for the AP Japanese Computer-Simulated Conversation Task: Evaluation Inference 153 task and addresses the scarcity of validity evidence for this exam. The findings underscore the importance of empirically  \nconfirming rating scale functioning in any assessment context.  \nKeywords: Japanese language testing; argument-based approach to validity; speaking assessment; simulated interactive conversation; AP Japanese exam  \nJNCOLCTL VOL 38  \n154 Suzumura-Smith  \nIntroduction  \nPerformance assessment involves a complex rating process. In this process, raters evaluate examinees’ performances based on their understanding of the target construct and scoring criteria (Eckes, 2015, 2019) . Raters and rating scales play important roles in linking examinees’ observed performances to their scores on assessment tasks (Kane, 2013) . Consequently, rater reliability and rating scale utility are critical components in building validity arguments for performance assessment (Eckes, 2015, 2019; Knoch, Deygers, et al., 2021; Knoch & Chapelle, 2018) .  \nThe importance of reliability in rater-mediated performance assessment has long been recognized (Eckes, 2015, 2019; Engelhard & Wind, 2018; Knoch, Fairbairn, et al., 2021) . Rater variability has been extensively studied using both quantitative and qualitative methods (Bachman et al., 1995; Brown et al., 2005; Knoch, Deygers, et al., 2021; Liu & Xie, 2014; Ma, 2022; Taguchi, 2011; Yan, 2014; Youn, 2018) . However, the utility of rating scales has received less attention  \nJNCOLCTL VOL 38  \nRater Reliability and Rating Scale Utility for the AP Japanese Computer-Simulated Conversation Task: Evaluation Inference 155 (Mendoza & Knoch, 2018) . For high-stakes performance testing, rating scales are typically developed by a panel of content experts based on research and existing standards (Kane et al., 2017) . This expert involvement is often presented as the sole support for rating scale utility. However, expert involvement alone is not sufficient evidence. It is not advisable to assume that rating scales function as intended. Empirical confirmation through field testing or operational use is necessary (Knoch & Chapelle, 2018) .  \nPrevious research shows that various factors can influence rating scale utility, including test taker characteristics, raters, speech acts, and proficiency levels (Li et al., 2019; Taguchi, 2011; Yan, 2014) . Therefore, it is essential to empirically investigate both rater reliability and rating scale utility within each assessment context. This paper examines these two aspects in the context of the computer-simulated conversation task on the Advanced Placement (AP) Japanese Language and Culture Exam (hereafter the AP Japanese exam) .  \nJNCOLCTL VOL 38  \n156 Suzumura-Smith  \nAP Japanese Language and Culture Exam  \nThe AP Japanese exam is a high-stakes criterion-referenced test developed by the College Board","cbCaig225wlZ29e4","https://ap.wps.com/l/cbCaig225wlZ29e4","pdf",1762081,79,"English","# Abstract\n# Introduction\n## Rater reliability and rating scale utility\n## Prior work and need for empirical evidence\n# AP Japanese Language and Culture Exam\n## Test purpose and structure\n## Computer-simulated conversation task","[{\"question\":\"What aspects of scoring validity were investigated for the AP Japanese conversation task?\",\"answer\":\"The study focused on rater reliability and the functioning of the rating scale, using an argument-based approach to validity.\"},{\"question\":\"How was the task administered and how were performances scored?\",\"answer\":\"A computer simulation collected responses from 102 high school students, and three raters scored performances using a common 7-point holistic scale.\"},{\"question\":\"What were the main results regarding rater reliability and rating scale utility?\",\"answer\":\"Results supported rater reliability, but they provided only limited support for how well the rating scale functioned as intended.\"}]","Rater Reliability and Rating Scale Utility for the AP Japanese Computer-Simulated Conversation Task - Evaluation Inference | PDF",1790784264,199]