[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-detail-184119-en":3,"doc-seo-184119-105":30,"detail-sidebar-cat-0-en-105":92},{"code":4,"msg":5,"data":6},0,"success",{"doc_id":7,"user_id":8,"nickname":9,"user_avatar":10,"doc_module":4,"category_id":11,"category_name":12,"doc_title":13,"doc_description":14,"doc_content":15,"file_id":16,"file_url":17,"file_type":18,"file_size":19,"view_count":20,"is_deleted":4,"is_public":20,"is_downloadable":20,"audit_status":20,"page_count":21,"language":22,"language_code":23,"site_id":24,"html_lang":23,"table_of_contents":25,"faqs":26,"seo_title":27,"seo_description":14,"update_tm":28,"read_time":29},184119,13056703020460,"Valentina","https://ap-avatar.wpscdn.com/avatar/be000253dac470eee5d?_k=1778207105932848923",8,"Research & Report","SLAM18 settles - token and AUC performance results","Model performance comparisons across English, Spanish, and French tracks evaluate training, development, and test token volumes alongside user counts. Results report Team AUC and F1 rankings for top teams and investigate the contribution of system components such as recurrent neural networks, decision tree ensembles, multitask models, and linear models. Additional ablation-style parameter effects are summarized via intercept, user/team/track ID variability, and feature-level factors including word corpus frequency, spaced repetition features, and embeddings, with language-specific oracle and stacking baselines.","| Track | Users | TRAIN\u003Cbr>Tokens (Err) | DEV\u003Cbr>Tokens (Err) | TEST\u003Cbr>Tokens (Err) |\n| --- | --- | --- | --- | --- |\n| English | 2.6k | 2.6M (13%) | 387k (14%) | 387k (15%) |\n| Spanish | 2.6k | 2.0M (14%) | 289k (16%) | 282k (16%) |\n| French | 1.2k | 927k (16%) | 138k (18%) | 136k (18%) |\n| Overall | 6.4k | 5.5M (14%) | 814k (15%) | 804k (16%) |\n\n|  | English Track |  |  | Spanish Track |  |  | French Track |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| ↑ | Team AUC | F1 | ↑ | Team AUC | F1 | ↑ | Team | AUC | F1 |\n| 1 | SanaLabs ♢♣ .861 | .561 | 1 | SanaLabs ♢♣ .838 | .530 | 1 | SanaLabs ♢♣ | .857 | .573 |\n| 1 | singsound ♢ .861 | .559 | 2 | NYU ♣‡ .835 | \u003Cbr>.420 | 2 | singsound ♢ | .854 | .569 |\n| 3 | NYU ♣‡ .859 | \u003Cbr>.468 | 2 | singsound ♢ .835 | \u003Cbr>.524 | 2 | NYU ♣‡ | .854 | .493 |\n| 4 | TMU ♢‡ .848 | .476 | 4 | TMU ♢‡ .824 | .439 | 4 | CECL ‡ | .843 | .487 |\n| 5 | CECL ‡ .846 | \u003Cbr>.414 | 5 | CECL ‡ .818 | \u003Cbr>.390 | 5 | TMU ♢‡ | .839 | .502 |\n| 6 | Cambridge ♢ .841 | .479 | 6 | Cambridge ♢ .807 | .435 | 6 | Cambridge ♢ | .835 | .508 |\n| 7 | UCSD ♣ .829 | \u003Cbr>.424 | 7 | UCSD ♣ .803 | \u003Cbr>.375 | 7 | UCSD ♣ | .823 | .442 |\n| 8 | nihalnayak .821 | .376 | 7 | LambdaLab ♣ .801 | \u003Cbr>.344 | 8 | LambdaLab ♣ | .815 | .415 |\n| 8 | LambdaLab ♣ .821 | .389 | 9 | Grotoco .791 | .452 | 8 | Grotoco | .813 | .502 |\n| 10 | Grotoco .817 | \u003Cbr>.462 | 9 | nihalnayak .790 | .338 | 10 | nihalnayak | .811 | .431 |\n| 11 | jilljenn .815 | .329 | 11 | ymatusevych .789 | \u003Cbr>.347 | 10 | jilljenn | .809 | .406 |\n| 12 | ymatusevych .813 | \u003Cbr>.381 | 11 | jilljenn .788 | \u003Cbr>.306 | 10 | ymatusevych | .808 | .441 |\n| 13 | renhk .797 | .448 | 13 | renhk .773 | .432 | 13 | simplelinear | .807 | .394 |\n| 14 | zlb241 .787 | \u003Cbr>.003 | 14 | SLAM_baseline .746 | \u003Cbr>.175 | 14 | renhk | .796 | .481 |\n| 15 |  SLAM_baseline .774  | \u003Cbr>.190 | 15 | zlb241 .682 | .389 | 15 | SLAM_baseline | .771 | .281 |\n\n\n| Intercept | .786 \u003C.001 *** |\n| --- | --- |\n| Recurrent neural network (♢) | +.028 .012 * |\n| Decision tree ensemble (♣) | +.018 .055 . |\n| Linear model (e.g., IRT) | −.006 .541 |\n| Multitask model (‡) | +.023 .017 * |\n| Random effects | St. Dev. |\n| User ID | ±.086 |\n| Team ID | ±.013 |\n| Track ID | ±.011 |\n\n\n| Word (surface form)\u003Cbr>\u003Cbr>User ID\u003Cbr>Part of speech  Dependency labels  Morphology features  Response time  Days in course\u003Cbr>\u003Cbr>Client | |  +.005\u003Cbr>  +.014 \u003Cbr> −.008\u003Cbr>\u003Cbr>−.011\u003Cbr>−.021\u003Cbr> +.028 * \u003Cbr>+.023 .\u003Cbr>\u003Cbr>+.005 |\n| --- | --- | --- |\n| Countries  |  | +.012 |\n| \u003Cbr>Dependency edges  |  | \u003Cbr>−.000 |\n| Session  |  | +.014 |\n| Word corpus frequency  |  | +.008 |\n| \u003Cbr>Spaced repetition features  |  | \u003Cbr>+.013 |\n| L1-L2 cognates  |  | +.001 |\n| \u003Cbr>Word embeddings  |  | \u003Cbr>+.020 |\n| Word stem/root/lemma  |  | +.007 |\n\n| System | English Spanish French |\n| --- | --- |\n| Oracle | .995 .996 .993 |\n| Stacking | .867 .844 .863 |\n| Average (top 3) | .867 .843 .863 |\n| 1st team | .861 .838 .857 |\n| 2nd team | .861 .835 .854 |\n| 3rd team | .859 .835 .854 |\n| \u003Cbr>Average (all) 4th team | \u003Cbr>.857 .832 .852\u003Cbr>.848 .824 .843 |","cbCaituYMNprhaMr","https://ap.wps.com/l/cbCaituYMNprhaMr","pdf",660301,1,10,"English","en",105,"# Track overview\n## Token volumes (TRAIN/DEV/TEST)\n## Team rankings by Track (AUC, F1)\n# System comparisons\n## Oracle and stacking results\n# Model effects and feature contributions\n## Coefficients for system components\n## Feature-level impact","[{\"question\":\"How are token volumes summarized for each language track?\",\"answer\":\"The document lists users and token counts for TRAIN, DEV, and TEST per language (English, Spanish, French) and an overall total, including error percentages.\"},{\"question\":\"What metrics are used to compare teams within each track?\",\"answer\":\"Team performance is reported using Team AUC and F1, with ranked listings for English, Spanish, and French tracks.\"},{\"question\":\"Which systems and features show measurable effects in the analysis?\",\"answer\":\"The document highlights effects for recurrent neural networks, decision tree ensembles, multitask models, and random effects, and it lists feature contributions such as spaced repetition features, word corpus frequency, and word embeddings.\"}]","SLAM18 settles - token and AUC performance results | PDF",1788356283,25,{"code":4,"msg":31,"data":32},"ok",{"site_id":24,"language":23,"slug":33,"title":13,"keywords":34,"description":14,"schema_data":35,"social_meta":87,"head_meta":89,"extra_data":91,"updated_unix":28},"slam18-settles-token-and-auc-performance-results","",{"@graph":36,"@context":86},[37,54,69],{"@type":38,"itemListElement":39},"BreadcrumbList",[40,44,48,51],{"item":41,"name":42,"@type":43,"position":20},"https://docshare.wps.com","Home","ListItem",{"item":45,"name":46,"@type":43,"position":47},"https://docshare.wps.com/document/","Document",2,{"item":49,"name":12,"@type":43,"position":50},"https://docshare.wps.com/document/research-report/",3,{"item":52,"name":13,"@type":43,"position":53},"https://docshare.wps.com/document/slam18-settles-token-and-auc-performance-results/184119/",4,{"url":52,"name":13,"@type":55,"author":56,"headline":13,"publisher":58,"fileFormat":61,"inLanguage":23,"description":14,"dateModified":62,"datePublished":63,"encodingFormat":61,"isAccessibleForFree":64,"interactionStatistic":65},"DigitalDocument",{"name":9,"@type":57},"Person",{"url":41,"name":59,"@type":60},"DocShare","Organization","application/pdf","2026-09-04","2026-09-02",true,{"@type":66,"interactionType":67,"userInteractionCount":20},"InteractionCounter",{"@type":68},"ViewAction",{"@type":70,"mainEntity":71},"FAQPage",[72,78,82],{"name":73,"@type":74,"acceptedAnswer":75},"How are token volumes summarized for each language track?","Question",{"text":76,"@type":77},"The document lists users and token counts for TRAIN, DEV, and TEST per language (English, Spanish, French) and an overall total, including error percentages.","Answer",{"name":79,"@type":74,"acceptedAnswer":80},"What metrics are used to compare teams within each track?",{"text":81,"@type":77},"Team performance is reported using Team AUC and F1, with ranked listings for English, Spanish, and French tracks.",{"name":83,"@type":74,"acceptedAnswer":84},"Which systems and features show measurable effects in the analysis?",{"text":85,"@type":77},"The document highlights effects for recurrent neural networks, decision tree ensembles, multitask models, and random effects, and it lists feature contributions such as spaced repetition features, word corpus frequency, and word embeddings.","https://schema.org",{"og:url":52,"og:type":88,"og:title":13,"og:site_name":59,"og:description":14},"article",{"robots":90,"canonical":52},"index,follow",{"doc_id":7,"site_id":24},{"code":4,"msg":5,"data":93},[94,98,102,106,111,116,121,124,129,132,135],{"id":20,"doc_module":4,"doc_module_name":46,"category_name":95,"show_sort_weight":96,"slug":97},"Story & Novel",90,"story-novel",{"id":47,"doc_module":4,"doc_module_name":46,"category_name":99,"show_sort_weight":100,"slug":101},"Literature",80,"literature",{"id":53,"doc_module":4,"doc_module_name":46,"category_name":103,"show_sort_weight":104,"slug":105},"Exam",70,"exam",{"id":107,"doc_module":4,"doc_module_name":46,"category_name":108,"show_sort_weight":109,"slug":110},5,"Comic",60,"comic",{"id":112,"doc_module":4,"doc_module_name":46,"category_name":113,"show_sort_weight":114,"slug":115},6,"Technology",50,"technology",{"id":117,"doc_module":4,"doc_module_name":46,"category_name":118,"show_sort_weight":119,"slug":120},7,"Healthcare",40,"healthcare",{"id":11,"doc_module":4,"doc_module_name":46,"category_name":12,"show_sort_weight":122,"slug":123},30,"research-report",{"id":125,"doc_module":4,"doc_module_name":46,"category_name":126,"show_sort_weight":127,"slug":128},9,"Religion & Spirituality",20,"religion-spirituality",{"id":127,"doc_module":4,"doc_module_name":46,"category_name":130,"show_sort_weight":127,"slug":131},"World Cup","world-cup",{"id":21,"doc_module":4,"doc_module_name":46,"category_name":133,"show_sort_weight":21,"slug":134},"Lifestyle","lifestyle",{"id":136,"doc_module":4,"doc_module_name":46,"category_name":137,"show_sort_weight":107,"slug":138},19,"General","general"]