[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-178729-105":59,"doc-detail-178729-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","cefr-labeling-metrics-and-test-taker-response-metrics","CEFR-labeling Metrics and Test-taker Response Metrics","","Material focuses on evaluating reading comprehension or language assessment models using CEFR-related labeling metrics and test-taker response metrics. It reports correlations such as Pearson’s r and Spearman’s rank, as well as loss-like and calibration measures including cross-entropy and residual standard deviation. Results compare baselines and multiple approaches (e.g., IRT-style variants and BERT-LLTM with objective strategies) and include test–retest reliability and item-type/total correlation. Additional passages illustrate mappings from sample text to SMECEFR and predicted CEFR levels with estimated predicted difficulty.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":20,"@type":76,"position":81},"https://docshare.wps.com/document/exam/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/cefr-labeling-metrics-and-test-taker-response-metrics/178729/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/cefr-labeling-metrics-and-test-taker-response-metrics/178729.png","ImageObject",300,407,{"name":92,"@type":93},"นรินทร์","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-09","2026-09-02",true,{"@type":102,"interactionType":103,"userInteractionCount":29},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"Which metrics are used to evaluate CEFR labeling performance?","Question",{"text":112,"@type":113},"The document lists Pearson’s r, Spearman’s rank correlation, and additional scoring-related quantities including item mean score versus predicted Pearson’s r.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does the document assess test-taker response behavior?",{"text":117,"@type":113},"It uses response-oriented metrics such as cross-entropy, residual standard deviation, item-type versus total Pearson’s r, and test–retest reliability.",{"name":119,"@type":110,"acceptedAnswer":120},"What do the example passages show?",{"text":121,"@type":113},"They provide short dialogue or expository samples along with SMECEFR level, predicted CEFR level, and a predicted difficulty value for each passage.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},178729,1788333169,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":19,"category_name":20,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":29,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},2336475104957,"https://ap-avatar.wpscdn.com/avatar/22000c4c6bd8a5076e1?x-image-process=image/resize,m_fixed,w_180,h_180&k=1787554080175789136","| Model | CEFR-labeling Metrics |  | Test-taker Response Metrics |  |  |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n|  | Pearson's r | Spearman's 􀀚 | Item Mean Score /\u003Cbr>Pred. Pearson's r | Cross\u003Cbr>entropy | Residual [st. dev](st. dev). | Item-Type / Total Pearson's r | Test–Retest\u003Cbr>Reliability |\n| Baselines |  |  |  |  |  |  |  |\n| ALL-SAME | 0.00 | 0.00 | 0.32 | 0.73 | 0.21 | 0.41 | 0.18 |\n| 2PL-IRT | N/Ay | N/Ay | 0.74 | 0.56 | 0.16 | 0.74 | 0.62 |\n| SETTLES-ET-AL | 0.81 | 0.75 | 0.43 | 0.91 | 0.20 | 0.66 | 0.53 |\n| BERT-LLTM |  |  |  |  |  |  |  |\n| CEFR OBJECTIVE | 0.84 | 0.77 | 0.45 | 0.88 | 0.28 | 0.70 | 0.51 |\n| TEST-TAKER OBJECTIVE | 0.73 | 0.49 | 0.66 | 0.55 | 0.16 | 0.75 | 0.63 |\n| JOINT OBJECTIVE | 0.82 | 0.76 | 0.62 | 0.55 | 0.17 | 0.75 | 0.62 |\n| Jump-starting New Items* |  |  |  |  |  |  |  |\n| BERT-LLTM |  |  |  |  |  |  |  |\n| TEST-TAKER OBJECTIVE | 0.74 | 0.56 | 0.59 | 0.53 | 0.16 | N/Az | N/Az |\n| JOINT OBJECTIVE | 0.80 | 0.73 | 0.52 | 0.54 | 0.17 | N/Az | N/Az |\n\n| Passage | SMECEFR level | Predicted CEFR level | Predicted\u003Cbr>difﬁculty |\n| --- | --- | --- | --- |\n| Tara: Do you want to go to the museum today? Billy: No, I don't like the museum very much. I want togo to the movie theater. Tara: I don't like any of the movies at the movie theater. Billy: OK, we'll go to the caf . Tara: OK! | A2 | A2 | 􀀀6:75 |\n| Since water is so important, you might wonder if you're drinking enough. There is no magic amount of water that kids need to drink every day. Usually, kids like to drink something with meals and should deﬁnitely drink when they are thirsty. But when it's warm out or you're exercising, you'll need more. Be sure to drink some extra water when you're out in warm weather, especially while playing sports or exercising. | B1/B2 | B2 | 􀀀4:04 |\n| The same as all eight of Connecticut's counties, there is no county government and no county seat. In Connecticut, towns are responsible for all local government activities, including ﬁre and rescue, snow removal and schools. In a few cases, neighboring towns will share some resources ( e.g., water, gas, etc. ). New London County is only a group of towns on a map. It has no governmental authority. | B2 | B2 | 􀀀2:98 |\n| Ariel University, formerly the College of Judea and Samaria, is the major Israeli institution of higher\u003Cbr>education in the West Bank. With close to 13,000 students, it is Israel's largest public college. The college was accredited in 1994 and awards bachelor's degrees in arts, sciences, technology, architecture and physical therapy. The school's current temporary status is that of a “university institution” conferred by the Israel Defense Forces, but it remains without university accreditation. | B2 | B2 | 􀀀2:40 |\n| The basic operation of a telephone involves sound waves being converted into electrical signals. These signals can then be sent over long distances from a device transmitting these signals at one end to a device receiving them at another. The original telephone system involved direct connections between two locations or parties. However, this was rapidly changed to a more ﬂexible system where a central ofﬁce would direct calls towards an intended receiver. | C1 | C1 | 􀀀1:06 |","cbCaiqpQU8ZqR4KG","https://ap.wps.com/l/cbCaiqpQU8ZqR4KG","pdf",3069035,17,"English","# CEFR-labeling Metrics\n## Correlation and calibration measures\n# Test-taker Response Metrics\n## Reliability and item-type correlations\n# Passage-level CEFR prediction examples\n## Predicted difficulty estimation","[{\"question\":\"Which metrics are used to evaluate CEFR labeling performance?\",\"answer\":\"The document lists Pearson’s r, Spearman’s rank correlation, and additional scoring-related quantities including item mean score versus predicted Pearson’s r.\"},{\"question\":\"How does the document assess test-taker response behavior?\",\"answer\":\"It uses response-oriented metrics such as cross-entropy, residual standard deviation, item-type versus total Pearson’s r, and test–retest reliability.\"},{\"question\":\"What do the example passages show?\",\"answer\":\"They provide short dialogue or expository samples along with SMECEFR level, predicted CEFR level, and a predicted difficulty value for each passage.\"}]","CEFR-labeling Metrics and Test-taker Response Metrics | PDF",43]