[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-136604-105":59,"doc-detail-136604-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","classer-cross-lingual-annotation-projection-enhanced-through-script-similarity-for-fine-grained-named-entity-recognition","CLASSER - Cross-lingual Annotation Projection enhanced through Script Similarity for Fine-grained Named Entity Recognition","","CLASSER introduces a cross-lingual annotation projection framework that uses script similarity to build fine-grained named entity recognition (FgNER) datasets for low-resource languages. It first projects annotations from high-resource NER datasets to target languages using source-to-target parallel corpora and a multilingual encoder-based projection tool, then refines projected labels by leveraging datasets in script-similar languages. Experiments on five low-resource Indian languages generate 1.8M sentences and show F1 gains of 26% for Marathi and 46% for Sanskrit, with additional zero-shot and cross-lingual analyses and public dataset release.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/classer-cross-lingual-annotation-projection-enhanced-through-script-similarity-for-fine-grained-named-entity-recognition/136604/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/classer-cross-lingual-annotation-projection-enhanced-through-script-similarity-for-fine-grained-named-entity-recognition/136604.png","ImageObject",300,407,{"name":92,"@type":93},"Emma Wilson","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-20","2026-08-22",true,{"@type":102,"interactionType":103,"userInteractionCount":44},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does CLASSER address?","Question",{"text":112,"@type":113},"CLASSER addresses the scarcity of high-quality fine-grained named entity recognition resources for low-resource languages, where manual annotation is costly and distant supervision is often noisy and concentrated in high-resource settings.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does CLASSER work in two stages?",{"text":117,"@type":113},"Stage 1 projects fine-grained annotations from a high-resource source language to the target language using parallel corpora and a multilingual encoder-based projection tool. Stage 2 refines the projected annotations using an auxiliary NER model trained on script-similar languages via a confidence-score strategy.",{"name":119,"@type":110,"acceptedAnswer":120},"Which languages are used to evaluate and build the CLASSER datasets?",{"text":121,"@type":113},"CLASSER is applied to five low-resource Indian languages: Assamese, Marathi, Nepali, Sanskrit, and Bodo.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},136604,1787389847,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":44,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":36},3848291630094,"https://eur-avatar.wpscdn.com/davatar_085a072bc5b1113ac321206ff7593b45","CLASSER: Cross-lingual Annotation Projection enhancement through Script Similarity for Fine-grained Named Entity Recognition  \nPrachuryya Kaushik and Ashish Anand  \nDepartment of Computer Science and Engineering  \nIndian Institute of Technology Guwahati  \nGuwahati, Assam, India  \n{k.prachuryya, [anand.ashish}@iitg.ac.in](anand.ashish}@iitg.ac.in)  \nAbstract  \nWe introduce CLASSER, a cross-lingual annotation projection framework enhanced through script similarity, to create fine-grained named entity recognition (FgNER) datasets for lowresource languages. Manual annotation for named entity recognition (NER) is expensive, and distant supervision often produces noisy data that are often limited to high-resource languages. CLASSER employs a two-stage process: first projection of annotations from high-resource NER datasets to target language by using source-to-target parallel corpora anda projection tool built on a multilingual encoder, then refining them by leveraging datasets in script-similar languages. We apply this to five low-resource Indian languages: Assamese, Marathi, Nepali, Sanskrit, and Bodo, a vulnerable language. The resulting dataset comprises  \n1.8M sentences, 2.6M entity mentions and  \n24.7M tokens. Through rigorous analyses, the effectiveness of our method and the high quality of the resulting dataset are ascertained with F1 score improvements of 26% in Marathi and 46% in Sanskrit over the current state-of-the-art. We further extend our analyses to zero-shot and cross-lingual settings, systematically investigating the impact of script similarity and multilingualism on cross-lingual FgNER performance. The dataset is publicly available at hugging[face.co/datasets/prachuryyaIITG/CLASSER](face.co/datasets/prachuryyaIITG/CLASSER).  \n1 Introduction  \nStructured knowledge extraction from unstructured text underpins countless downstream applications, such as recommendation systems, knowledgebase construction, relation extraction, and beyond. Named Entity Recognition (NER), which identifies and classifies mentions of persons, locations, organizations, etc. has evolved from early rule-based systems (Rau, 1991) through the collective contributions in the dedicated events (Grishman and Sundheim, 1996 ; Chinchor et al., 1998 ; Satoshi,  \nFgNER dataset in Parallel Corpora with sentence pairs  \nSource Language from Source to Target language  \nRefined Annotation in Target language for final dataset  \nFramework  \nFigure 1: Illustration of CLASSER Framework  \n2000 ; Tjong Kim Sang, 2002 ; Doddington et al., 2004 ; Santos et al., 2006) to powerful neural architectures today (Zhang et al., 2019 ; Zhou et al., 2023) . Yet, conventional coarse-grained categories in NER often fall short when applications demand more specific distinctions, e.g.,“Scientist” from generic “Person” or “Clothing” from generic “Product”(Choi et al., 2018) . The type and granularity of fine-grained entities differ depending on the domain and application requirements. Early efforts in fine-grained named entity recognition (FgNER) contributed hierarchical type systems (Sekine and Nobata, 2004), distant-supervision pipelines (Ling and Weld, 2012 ; Yosef et al., 2012), contextual embedding techniques (Gillick et al., 2014), and noise-aware neural architectures capable of predicting hundreds of labels (Murty et al., 2017) . Although noise is reduced in FgNER resources by applying language-specific heuristics (Abhishek et al., 2019), expensive manual annotation yields higher reliability and improved annotation quality (Ding et al., 2021) .  \nWhile coarse-grained NER for Indian languages  \n1745  \nProceedings of the 14th International Joint Conference on Natural Language Processing and the 4th Conference of the Asia-Pacific Chapter of the Association for  \nComputational Linguistics, pages 1745–1760  \nDecember 20-24, 2025 ©2025 Association for Computational Linguistics  \nhas seen considerable progress, fine-grained NER (FgNER) only began to emerge recently. The MultiCoNE","cbCaihVTqvIioS2I","https://ap.wps.com/l/cbCaihVTqvIioS2I","pdf",7436713,16,"English","# Introduction\n## Problem and Motivation\n## Proposed Solution: CLASSER Framework\n## Stage 1: Annotation Projection\n## Stage 2: Script-similarity Refinement","[{\"question\":\"What problem does CLASSER address?\",\"answer\":\"CLASSER addresses the scarcity of high-quality fine-grained named entity recognition resources for low-resource languages, where manual annotation is costly and distant supervision is often noisy and concentrated in high-resource settings.\"},{\"question\":\"How does CLASSER work in two stages?\",\"answer\":\"Stage 1 projects fine-grained annotations from a high-resource source language to the target language using parallel corpora and a multilingual encoder-based projection tool. Stage 2 refines the projected annotations using an auxiliary NER model trained on script-similar languages via a confidence-score strategy.\"},{\"question\":\"Which languages are used to evaluate and build the CLASSER datasets?\",\"answer\":\"CLASSER is applied to five low-resource Indian languages: Assamese, Marathi, Nepali, Sanskrit, and Bodo.\"}]","CLASSER - Cross-lingual Annotation Projection enhanced through Script Similarity for Fine-grained Named Entity Recognition | PDF"]