[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-443979-105":59,"doc-detail-443979-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","machine-learning-models-for-prediction-of-procathepsinglycosaminoglycan-binding-free-energies-based-on-molecular-structure","Machine learning models for prediction of (Pro)cathepsin–glycosaminoglycan binding free energies based on molecular structure","","Cathepsins are lysosomal and extracellular proteolytic enzymes whose activity depends on (pro)cathepsin state and modulation by negatively charged glycosaminoglycans (GAGs). The study builds machine learning models to predict MM-GBSA binding free energies for (pro)cathepsin–GAG complexes. Molecular dynamics simulations were run for six (pro)cathepsins and six GAGs, generating structural and energetic descriptors for eight ML algorithms. The fully connected neural network produced the most accurate predictions, and adding LIE components improved performance, enabling rapid structure-based screening.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/machine-learning-models-for-prediction-of-procathepsinglycosaminoglycan-binding-free-energies-based-on-molecular-structure/443979/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/machine-learning-models-for-prediction-of-procathepsinglycosaminoglycan-binding-free-energies-based-on-molecular-structure/443979.png","ImageObject",300,407,{"name":92,"@type":93},"Sarah ","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-04","2026-09-29",true,{"@type":102,"interactionType":103,"userInteractionCount":14},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is the goal of the machine learning models in this study?","Question",{"text":112,"@type":113},"To predict MM-GBSA binding free energies for (pro)cathepsin–GAG complexes using descriptors derived from molecular dynamics simulations.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How were the training inputs generated?",{"text":117,"@type":113},"The study used molecular dynamics simulations with ff14SB/GLYCAM06j to compute structural and energetic descriptors for multiple protein–GAG states and binding poses.",{"name":119,"@type":110,"acceptedAnswer":120},"Which ML method achieved the best predictive accuracy?",{"text":121,"@type":113},"The fully connected neural network (FCNN) yielded the most accurate predictions, with GradientBoost-based models performing comparably.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},443979,1791084485,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":14,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":145},962085320529,"https://ap-avatar.wpscdn.com/davatar_9964176cb1d06d4a9deccf72a44ae3dc","Computational and Structural Biotechnology Journal 31 (2026) 61–73  \nContents lists available at ScienceDirect  \nComputational and Structural Biotechnology Journal  \njournal [homepage:](homepage: www. elsevier. com/locate/cs bj)[ www. elsevier. com/locate/cs bj](homepage: www. elsevier. com/locate/cs bj)  \n| Research article\u003Cbr>Machine learning models for prediction of (Pro)cathepsin–glycosaminoglycan binding free energies based on molecular structure\u003Cbr>Krzysztof K. Bojarskia, b,∗ iD, Patrick K. Quoika b iD, Martin Zacharias ba Department of Physical Chemistry, Gdansk University of Technology, Narutowicza 11/12, Gdansk, Poland\u003Cbr>b Center for Functional Protein Assemblies, Technical University of Munich, Ernst-Otto-Fischer-Straße 8, Garching, Germany | |\n| --- | --- |\n| G R A P H I C A L A B S T R A C T |  |\n| |  |\n\nA R T I C L E I N F O  \nKeywords:  \nGlycosaminoglycans (GAGs)  \nMolecular dynamics (MD)  \nMachine learning (ML)  \nBinding energy calculation (MM-GBSA) Neural networks (NN)  \nQuantitative structure-activity relationship (QSAR)  \nA B S T R A C T  \nCathepsins are papain-like proteolytic enzymes localized in lysosomes and the extracellular matrix, where they participate in diverse physiological and pathological processes. They are synthesized as inactive precursors—procathepsins—containing a propeptide domain that blocks access to the active site. The activity of (pro)cathep sins can be modulated by glycosaminoglycans (GAGs), which are negatively charged, sulfated polysaccharides. This study aimed to develop machine learning (ML) models to predict MM-GBSA binding free energies in (pro)cathepsin–GAG complexes. Molecular dynamics simulations were performed using the ff14SB/GLYCAM06j force field for six (pro)cathepsins and six GAGs, representing four periodic states and six binding poses. Structural and energetic descriptors derived from these simulations were used as input features for eight ML algorithms: ElasticNet, Linear Regression, LinearSVR (with RBFSampler), LightGBM, Histogram Gradient Boosting, Fully Connected Neural Network (FCNN), and Random Forest. The FCNN yielded the most accurate predictions (􀁒2 = 0.7124 ± 0.0089; MAE = 5.2033 ± 0.0876 kcal/mol), with GradientBoost-based models performing comparably. Optimal FCNN performance was achieved with a minimal architecture (no hidden layers, dropout rate 0.01, ReLU activation). Incorporating Linear Interaction Energy (LIE) components significantly improved prediction accuracy, and approximately 17,000 data points were sufficient for stable model performance. Overall, this study provides a proof of concept for using ML to estimate binding free energies in protein–GAG systems and  \n∗ Corresponding author at: Department of Physical Chemistry, Gdansk University of Technology, Narutowicza 11/12, Gdansk, Poland.  \n[Email address:](Email address: krzysztof.bojarski@pg.edu.pl)[ krzysztof.bojarski@pg.edu.pl](Email address: krzysztof.bojarski@pg.edu.pl) (K.K. Bojarski).  \n[https://doi.org/10.1016/j.csbj.2025.11.059](https://doi.org/10.1016/j.csbj.2025.11.059)  \nReceived 7 October 2025; Received in revised form 26 November 2025; Accepted 26 November 2025  \nAvailable online 8 December 2025  \n2001-0370/© 2025 The Authors. Published by Elsevier B.V. on behalf of Research Network of Computational and Structural Biotechnology. This is an open access article under the CC BY license ([http://creativecommons.org/licenses/by/4.0/](http://creativecommons.org/licenses/by/4.0/)).  \nK.K. Bojarski, P.K. Quoika and M. Zacharias  \nComputational and Structural Biotechnology Journal 31 (2026) 61–73  \nestablishes a foundation for generalizable, structure-based predictors applicable to a broad range of biomolecular complexes. Beyond predictive accuracy, this approach enables rapid screening of MMGBSA interactions, facilitating the identification of favorable binding regions and accelerating structure-guided design efforts.  \n1. Introduction  \nGlycosaminoglycans (GAGs), negatively charged, unbr","cbCaieEKH0vH1nfS","https://ap.wps.com/l/cbCaieEKH0vH1nfS","pdf",4167666,13,"English","# Introduction\n## Background on cathepsins and GAG modulation\n## Challenges in modeling GAG-containing systems","[{\"question\":\"What is the goal of the machine learning models in this study?\",\"answer\":\"To predict MM-GBSA binding free energies for (pro)cathepsin–GAG complexes using descriptors derived from molecular dynamics simulations.\"},{\"question\":\"How were the training inputs generated?\",\"answer\":\"The study used molecular dynamics simulations with ff14SB/GLYCAM06j to compute structural and energetic descriptors for multiple protein–GAG states and binding poses.\"},{\"question\":\"Which ML method achieved the best predictive accuracy?\",\"answer\":\"The fully connected neural network (FCNN) yielded the most accurate predictions, with GradientBoost-based models performing comparably.\"}]","Machine learning models for prediction of (Pro)cathepsin–glycosaminoglycan binding free energies based on molecular structure | PDF",1790706276,33]