[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-128826-105":59,"doc-detail-128826-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","machine-learning-methods-of-property-prediction-in-drug-discovery-doctoral-thesis","Machine learning methods of property prediction in drug discovery - Doctoral thesis","","Molecular properties are critical in drug discovery, where binding affinity strongly influences hit and lead selection. Existing machine learning approaches can be too slow for large-scale screening or limited in generalizability. This thesis develops faster models for binding affinity prediction, improving upon voxel-based approaches and evaluating 2D versus 3D methods through extensive benchmarks. It also investigates supervised and unsupervised pretraining gains, shows how 2D-3D model combinations enable state-of-the-art active learning results, and introduces techniques for interpreting neural network predictions by locating key input-space regions. Correct protonation states and robust 3D graph-based models are further addressed, including micro-pKa estimation.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/machine-learning-methods-of-property-prediction-in-drug-discovery-doctoral-thesis/128826/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/machine-learning-methods-of-property-prediction-in-drug-discovery-doctoral-thesis/128826.png","ImageObject",300,407,{"name":92,"@type":93},"Maeve","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-20","2026-08-06",true,{"@type":102,"interactionType":103,"userInteractionCount":44},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"Why is binding affinity important in drug discovery?","Question",{"text":112,"@type":113},"Binding affinity is crucial for selecting hit and lead candidates because it helps predict molecular potency against a target of interest.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"What limitations do current machine learning models have?",{"text":117,"@type":113},"Existing models can lack speed for large-scale screening and may not generalize well across diverse scenarios.",{"name":119,"@type":110,"acceptedAnswer":120},"How does the thesis improve property prediction performance?",{"text":121,"@type":113},"It develops faster binding affinity prediction models, compares 2D and 3D architectures with extensive benchmarks, and shows that combining 2D and 3D models can achieve state-of-the-art results in active learning.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},128826,1786003726,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":44,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},2336474466712,"https://ap-avatar.wpscdn.com/davatar_a8503ba1806abce46bf441b54a3ca4cd","“output” — 2024/7/16 — 11:03 — page i — \\#1  \nMachine learning methods of property prediction in drug discovery  \nNikolai Schapin  \nTESI DOCTORAL UPF / year 2024  \nDirectors de la tesi  \nDr. Gianni De Fabritiis / Dr. Francesc de Paula Sabanes Zariquiey Department of Medicine and Life Sciences / Acellera Labs  \n“output” — 2024/7/16 — 11:03 — page ii — \\#2  \n“output” — 2024/7/16 — 11:03 — page iii — \\#3  \nTo those who follow their path and are not afraid to challenge existing beliefs or others’ opinions  \niii  \n“output” — 2024/7/16 — 11:03 — page iv — \\#4  \n“output” — 2024/7/16 — 11:03 — page v — \\#5  \nThanks  \nI would like to thank all my colleagues at Acellera Labs, the Computational Science Laboratory at UPF, my industrial supervisor at Acellera Francesc de Paula Sabanes Zariquiey and former colleagues Maciej Majewski, Raimondas Galvelisand Roberto Fino for their great support and friendship. Thanks to them, I have learned a lot in the fields of computational chemistry and structural biology. I would also like to thank my tutor and supervisor Gianni De Fabritiis for the opportunity to perform this doctoral project and for his guidance and teachings throughout the project.  \nAdditionally I would also like to thank Karla and whole my family for their emotional support throughout this time.  \nv  \n“output” — 2024/7/16 — 11:03 — page vi — \\#6  \n“output” — 2024/7/16 — 11:03 — page vii — \\#7  \nAbstract  \nMolecular properties are highly important in drug discovery with binding affinity being crucial for selecting hit and lead candidates. Existing machine learning models often lack speed for large-scale screening or lack generalizability. This thesis develops faster models for binding affinity prediction, surpassing previous voxel-based models. It compares 2D and 3D models through extensive benchmarks, highlighting scenarios where simpler 2D models outperform complex 3D models and showing how neural networks improve with supervised and unsupervised pretraining. The work also shows how combining 2D and 3D models yields state-of-the-art results in active learning. To address the difficult interpretability of neural network predictions, a new application is introduced to identify key contributing areas of the input space. Additionally, correct protonation states are essential for preparing molecular libraries. The thesis extends 3D graph-based models for micro-pKa estimation, presenting a new model and application that is more robust to diverse chemical compounds than existing methods.  \nResum  \nLes propietats moleculars sn molt importants en el descobriment de frmacs, on l’afinitat d’uni s crucial per seleccionar candidats potencials i caps de srie. Els models d’aprenentatge automtic existents sovint els hi falta velocitat suficient per a cribatges de gran escala o generalitzabilitat. Aquesta tesi desenvolupa models ms rpids per a la predicci d’afinitat d’uni, superant els models anteriors basats en vxels. La tesi compara models 2D i 3D a travs de d’extenses dades de referncia, remarcant escenaris on models 2D ms simples obtenen millors resultats que models 3D ms complexes i ensenyant com les xarxes neuronals milloren amb pre-entrenaments supervisats i no supervisats. El treball tamb mostra com la combinaci de models 2D i 3D produeix resultats puntersen aprenentatge actiu. Per abordar la dif´ıcil interpretabilitat de les prediccions de les xarxes neuronals, s’introdueix una nova aplicaci per identificar les rees clau que contribueixen a l’espai d’entrada. Addicionalment, la correcta protonaci de les molecules s essencial per a la preparaci de llibreries moleculars. Latesi estn els models 3D basats en grafs per a l’estimaci de micro-pKa, presentat un nou model i aplicaci ms robust envers la diversitat qu´ımica dels compostos que els mtodes actuals.  \nvii  \n“output” — 2024/7/16 — 11:03 — page viii — \\#8  \n“output” — 2024/7/16 — 11:03 — page ix — \\#9  \nPreface  \nEarly stage, preclinical drug discovery is characterized by a step-wise proces","cbCaifHuqlJ2KVsH","https://ap.wps.com/l/cbCaifHuqlJ2KVsH","pdf",21305013,294,"English","# Abstract\n## Binding affinity prediction and speed vs generalizability\n## 2D vs 3D models and pretraining\n## Active learning using combined 2D/3D models\n## Interpretability via key input-space areas\n## Protonation states and micro-pKa estimation","[{\"question\":\"Why is binding affinity important in drug discovery?\",\"answer\":\"Binding affinity is crucial for selecting hit and lead candidates because it helps predict molecular potency against a target of interest.\"},{\"question\":\"What limitations do current machine learning models have?\",\"answer\":\"Existing models can lack speed for large-scale screening and may not generalize well across diverse scenarios.\"},{\"question\":\"How does the thesis improve property prediction performance?\",\"answer\":\"It develops faster binding affinity prediction models, compares 2D and 3D architectures with extensive benchmarks, and shows that combining 2D and 3D models can achieve state-of-the-art results in active learning.\"}]","Machine learning methods of property prediction in drug discovery - Doctoral thesis | PDF",741]