[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-140919-105":59,"doc-detail-140919-en":131},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":124,"head_meta":126,"extra_data":128,"updated_unix":130},105,"en","evaluating-subword-tokenization-alien-subword-composition-and-oov-generalization-challenge","Evaluating Subword Tokenization - Alien Subword Composition and OOV Generalization Challenge","","Popular subword tokenizers used in language models, such as Byte-Pair Encoding (BPE), often ignore morpheme boundaries, which harms downstream performance. To address cross-tokenizer evaluation, a combined intrinsic–extrinsic framework is proposed. Intrinsic scoring uses UniMorph Labeller (umLabeller) to classify subword compositions as morphological or alien, while extrinsic testing uses the OOV Generalization Challenge 1.0 with three downstream text classification tasks under covariate shifts. Results show umLabeller reaches 98% accuracy and alien tokenization generalizes worse than morphological tokenization across ALBERT, BERT, RoBERTa, and DeBERTa.",{"@graph":69,"@context":123},[70,84,106],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/evaluating-subword-tokenization-alien-subword-composition-and-oov-generalization-challenge/140919/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/evaluating-subword-tokenization-alien-subword-composition-and-oov-generalization-challenge/140919.png","ImageObject",300,407,{"name":92,"@type":93},"Caleb Sterling","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-16","2026-08-25",true,{"@type":102,"interactionType":103,"userInteractionCount":105},"InteractionCounter",{"@type":104},"ViewAction",11,{"@type":107,"mainEntity":108},"FAQPage",[109,115,119],{"name":110,"@type":111,"acceptedAnswer":112},"为什么现有子词标记化方法会影响模型下游性能？","Question",{"text":113,"@type":114},"因为主流子词标记器（如 BPE）往往不尊重词素边界，导致切分结果与人类理解的形态语义组合不一致，从而影响下游任务表现。","Answer",{"name":116,"@type":111,"acceptedAnswer":117},"umLabeller 的 intrinsic evaluation 如何工作？",{"text":118,"@type":114},"umLabeller 会对给定的子词切分进行判断，将其归类为形态学（morphological）或“外星”（alien）组合，用于表征子词级语义组合的合理性。",{"name":120,"@type":111,"acceptedAnswer":121},"OOV Generalization Challenge 1.0 benchmark 用于评估什么？",{"text":122,"@type":114},"该基准包含三个下游文本分类子任务，在微调与测试阶段之间构造完全生成的协变量转移，评估语言模型的组合泛化与形态泛化能力，并用 umLabeller 的输出来驱动挑战。","https://schema.org",{"og:url":83,"og:type":125,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":127,"canonical":83},"index,follow",{"doc_id":129,"site_id":62},140919,1787646154,{"code":4,"msg":5,"data":132},{"doc_id":129,"user_id":133,"nickname":92,"user_avatar":134,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":135,"file_id":136,"file_url":137,"file_type":138,"file_size":139,"view_count":105,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":140,"language":141,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":142,"faqs":143,"seo_title":144,"seo_description":67,"update_tm":130,"read_time":145},962084925290,"https://ap-avatar.wpscdn.com/davatar_085a072bc5b1113ac321206ff7593b45","Evaluating Subword Tokenization:  \nAlien Subword Composition and OOV Generalization Challenge  \nKhuyagbaatar Batsuren 1 ,2 , Ekaterina Vylomova 1 , Verna Dankers3 , Tsetsuukhei Delgerbaatar2 , Omri Uzan4 , Yuval Pinter4 , and Gábor Bella5  \n1University of Melbourne 2National University of Mongolia 3University of Edinburgh  \n4Ben-Gurion University of the Negev 5IMT Atlantique  \n[khuyagbaatar.b@gmail.com](khuyagbaatar.b@gmail.com)  \narXiv :2404 . 13292v1 [ cs .CL] 20 Apr 2024  \nAbstract  \nThe popular subword tokenizers of current language models, such as Byte-Pair Encoding (BPE), are known not to respect morpheme boundaries, which affects the downstream performance of the models. While many improved tokenization algorithms have been proposed, their evaluation and cross-comparison are still an open problem. As a solution, we propose a combined intrinsic–extrinsic evaluation framework for subword tokenization. Intrinsic evaluation is based on our new UniMorph Labeller (umLabeller) tool that classifies a subword tokenization as either morphological or alien. Extrinsic evaluation, in turn, is performed via the Out-of-Vocabulary Generalization Challenge 1.0 benchmark, which consists of three newly specified downstream text classification tasks. Our empirical findings show that the accuracy of umLabeller is 98%, and that, in all language models studied (including ALBERT, BERT, RoBERTa, and DeBERTa), alien tokenization leads to poorer generalizations compared to morphological tokenization for semantic compositionality of word meanings.  \n1 Introduction  \nSubword tokenization is a fundamental preprocessing method in Natural Language Processing that segments words into subword units. Popular subword tokenization methods, such as Byte Pair Encoding (BPE; Sennrich et al., 2016) or Unigram Language Model (ULM; Kudo, 2018), are adaptations of data compression algorithms that mainly rely on word character co-occurrence statistics in a given text corpus, rather than on human knowledge and understanding about word formations and morphology. As a result, certain subword compositions produced by these tokenizers are not aligned with any semantic compositions as understood by humans (e.g., h _iked in the GPT-4 Tokenizer) . In therest of the paper, we refer to such linguistically implausible subword compositions as alien composi-  \njogging hiked immunizers  \nTokenizer  \n GPT-3   ALBERT   OPT   \nj _ogging h _ikedimmun _izers  \nUniMorph  \numLabeller  \nalien  \nalien morph  \nFigure 1: umLabeller, the inspection tool for characterizing semantic compositionality of the subword-level tokenization into morph and alien.  \ntions. In alien subword compositions, subwords are not recognized by us humans as meaningful units from which the overall word meaning is composed. For example, in case of j _ogging (segmentation produced by GPT-3 and RoBERTa tokenizers), no subword in this composition represents a meaning of jog. This issue is well illustrated by adversarial attacks, as shown in Table 1, to which current language models are vulnerable.  \nDespite the success of subword tokenization in popular NLP applications (including machine translation, text generation, and text classification) anda wide range of practical (Mielke et al., 2021) and cognitive studies (Beinborn and Pinter, 2023), evaluating subword tokenization algorithms is still an open problem for at least two reasons. Firstly, stateof-the-art evaluations in NLP models lack a unified set of criteria, as well as the underlying decision process, to verify the intrinsic correctness of a given subword tokenization. The second motivation is that little or no effort has gone into developing a standard extrinsic NLP benchmark to evaluate how tokenizers with different behaviors impact predictions of downstream tasks in NLP (Truong et al., 2024) .  \nIn this work, we first describe umLabeller, a large-scale, high-quality characterization algorithm and tool for subword compositions. umLabeller is well adapted to","cbCaio1dKOMzwOwv","https://ap.wps.com/l/cbCaio1dKOMzwOwv","pdf",487750,13,"English","# Introduction\n## Proposed intrinsic evaluation with umLabeller\n## Proposed extrinsic benchmark: OOV Generalization Challenge 1.0","[{\"question\":\"为什么现有子词标记化方法会影响模型下游性能？\",\"answer\":\"因为主流子词标记器（如 BPE）往往不尊重词素边界，导致切分结果与人类理解的形态语义组合不一致，从而影响下游任务表现。\"},{\"question\":\"umLabeller 的 intrinsic evaluation 如何工作？\",\"answer\":\"umLabeller 会对给定的子词切分进行判断，将其归类为形态学（morphological）或“外星”（alien）组合，用于表征子词级语义组合的合理性。\"},{\"question\":\"OOV Generalization Challenge 1.0 benchmark 用于评估什么？\",\"answer\":\"该基准包含三个下游文本分类子任务，在微调与测试阶段之间构造完全生成的协变量转移，评估语言模型的组合泛化与形态泛化能力，并用 umLabeller 的输出来驱动挑战。\"}]","Evaluating Subword Tokenization - Alien Subword Composition and OOV Generalization Challenge | PDF",33]