[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-457659-105":59,"doc-detail-457659-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","optimizing-large-language-models-for-ontology-based-annotation-a-study-on-gene-ontology-in-biomedical-texts","Optimizing large language models for ontology-based annotation - a study on gene ontology in biomedical texts","","Automated ontology annotation of scientific literature supports knowledge management in biology and biomedicine by enabling accurate concept tagging for information retrieval, semantic search, and knowledge integration. The study evaluates large language models—including MPT-7B, Phi, BiomedLM, and Meditron—for Gene Ontology (GO) concepts. Models are fine-tuned on the CRAFT dataset and assessed using F1 score, semantic similarity, memory usage, and inference speed. Results indicate competitive Bi-GRU accuracy and qualitatively higher semantic consistency for some complex ontology terms. High resource requirements are addressed via PEFT and advanced prompting, balancing annotation gains with computational cost.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/optimizing-large-language-models-for-ontology-based-annotation-a-study-on-gene-ontology-in-biomedical-texts/457659/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/optimizing-large-language-models-for-ontology-based-annotation-a-study-on-gene-ontology-in-biomedical-texts/457659.png","ImageObject",300,407,{"name":92,"@type":93},"SANS","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-08","2026-09-30",true,{"@type":102,"interactionType":103,"userInteractionCount":34},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does the study address in scientific knowledge management?","Question",{"text":112,"@type":113},"It addresses automated ontology annotation of scientific literature, which improves knowledge management through accurate concept tagging for information retrieval, semantic search, and knowledge integration.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"Which large language models and dataset are used for ontology annotation in the study?",{"text":117,"@type":113},"The study evaluates large language models such as MPT-7B, Phi, BiomedLM, and Meditron, fine-tuned on the CRAFT dataset for Gene Ontology (GO) concepts.",{"name":119,"@type":110,"acceptedAnswer":120},"How are the models evaluated and what key results are reported?",{"text":121,"@type":113},"Performance is measured using F1 score, semantic similarity, memory usage, and inference speed. Bi-GRU baselines remain competitive for raw accuracy, while LLMs show qualitatively higher semantic consistency for some complex or multi-word ontology terms.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},457659,1791305738,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":34,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":145},962090760505,"https://ap-avatar.wpscdn.com/davatar_155a257f0dc6eb9ab79c44ca47cae57d","Devkota et al. BioData Mining (2026) 19:5 [https://doi.org/10.1186/s13040-025-00507-z](https://doi.org/10.1186/s13040-025-00507-z)  \nBioData Mining  \nRESEARCH Open Access  \nOptimizing large language models  \nfor ontology-based annotation: a study on gene ontology in biomedical texts  \nPratik Devkota 1, Somya D. Mohanty2 and Prashanti Manda3*  \n*Correspondence:  \nPrashanti Manda[pmanda@unomaha.edu](pmanda@unomaha.edu)[ ](pmanda@unomaha.edu)1Informatics and Analytics, University of North Carolina Greensboro, Forest St, Greensboro, NC 27455, USA  \n2Artificial Intelligence, UnitedHealth Group, Minneapolis, MN  \n55440, USA  \n3Department of Computer Science, University of Nebraska Omaha, S. 67th St, Omaha, NE 68182, USA  \nAbstract  \nAutomated ontology annotation of scientific literature plays a critical role in knowledge management, particularly in fields like biology and biomedicine, where accurate concept tagging can enhance information retrieval, semantic search, and knowledge integration. Traditional models for ontology annotation, such as Recurrent Neural Networks (RNNs) and Bidirectional Gated Recurrent Units (Bi-GRUs), have been effective but limited in handling complex biomedical terminologies and semantic nuances. This study explores the potential of large language models (LLMs), including MPT-7B, Phi, BiomedLM, and Meditron, for improving ontology annotation, specifically with Gene Ontology (GO) concepts. We fine-tuned these model s on the CRAFT dataset, assessing their performance in terms of F1 score, semantic similarity, memory usage, and inference speed. Our results show that while Bi-GRU baselines remain competitive in raw accuracy, LLMs offer complementary strengths. LLMs exhibit qualitatively higher semantic consistency in some cases, particularly when handling complex or multi-word ontology terms. However, these observations are exploratory and not statistically verified across all model types. However, resource requirements for LLMs are notably high, raising considerations about computational efficiency. Techniques like parameter-efficient fine-tuning (PEFT) and advanced prompting were explored to address these challenges, demonstrating potential in reducing computational demands while maintaining performance. Our findings suggest that while LLMs offer advantages in annotation accuracy, practical deployment should balance these benefits with resource costs. This research highlights the need for further optimization and domain-specific training to make LLMs a feasible choice for real-world biomedical ontology annotation tasks.  \nKeywords large language models, gene ontology, automated annotation, natural language processing  \nIntroduction  \nAutomatically annotating scientific literature with concepts from a domain ontology isan important task for several fields, especially Biology and biomedical sciences. Automated ontology annotation of literature refers to the process of automatically tagging  \n© The Author(s) 2025. Open Access This article is licensed under a Creative Commons Attribution-NonCommercial-NoDerivatives 4.0 International License, which permits any non-commercial use, sharing, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if you modified the licensed material. You do not have permission under this licence to share adapted material derived from this article or parts of it. The images or other third party material in this article are included in the article’s Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article’s Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit [http://creativecommons.org/l](http://creati","cbCaimAXFYMiRg6R","https://ap.wps.com/l/cbCaimAXFYMiRg6R","pdf",2929759,23,"English","# Abstract\n# Introduction\n## Key steps in automated ontology annotation\n## Traditional approaches and machine learning methods","[{\"question\":\"What problem does the study address in scientific knowledge management?\",\"answer\":\"It addresses automated ontology annotation of scientific literature, which improves knowledge management through accurate concept tagging for information retrieval, semantic search, and knowledge integration.\"},{\"question\":\"Which large language models and dataset are used for ontology annotation in the study?\",\"answer\":\"The study evaluates large language models such as MPT-7B, Phi, BiomedLM, and Meditron, fine-tuned on the CRAFT dataset for Gene Ontology (GO) concepts.\"},{\"question\":\"How are the models evaluated and what key results are reported?\",\"answer\":\"Performance is measured using F1 score, semantic similarity, memory usage, and inference speed. Bi-GRU baselines remain competitive for raw accuracy, while LLMs show qualitatively higher semantic consistency for some complex or multi-word ontology terms.\"}]","Optimizing large language models for ontology-based annotation - a study on gene ontology in biomedical texts | PDF",1790750089,58]