[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-349193-105":59,"doc-detail-349193-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","dna-methylation-biomarkers-based-pan-cancer-classifier-predictive-modeling-for-cancer-classification","DNA methylation biomarkers-based pan-cancer classifier - predictive modeling for cancer classification","","Machine-learning–driven molecular diagnostics using omics data can improve personalized medicine, yet translating models into diagnostic protocols is blocked by methodological pitfalls that overestimate performance during development and fail in real deployment. This study develops and validates a pan-cancer classification framework based on DNA methylation data, incorporating controlled biomarker selection, nested cross-validation evaluation, and anomaly filtering to address omics-data ML challenges and improve robustness for cancer diagnosis.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/dna-methylation-biomarkers-based-pan-cancer-classifier-predictive-modeling-for-cancer-classification/349193/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/dna-methylation-biomarkers-based-pan-cancer-classifier-predictive-modeling-for-cancer-classification/349193.png","ImageObject",300,407,{"name":92,"@type":93},"Cart","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-24","2026-09-22",true,{"@type":102,"interactionType":103,"userInteractionCount":14},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What problem does the study address in machine-learning molecular diagnostics?","Question",{"text":112,"@type":113},"It targets methodological challenges that often cause inflated model performance during development and poor performance when models are implemented in real-world diagnostic settings.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How is the pan-cancer classification framework developed and validated?",{"text":117,"@type":113},"The study curates a primary DNA methylation dataset and a validation dataset from independent studies, builds a custom biomarker selection strategy using an effect-size metric, and evaluates models using nested cross-validation.",{"name":119,"@type":110,"acceptedAnswer":120},"Why does the study include anomaly filtering in the inference pipeline?",{"text":121,"@type":113},"Local outlier factor is used to identify and filter samples with technical or biological anomalies, which improves classification performance across tested sample categories.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},349193,1790252510,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":14,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":56,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":143,"read_time":144},18829141979164,"https://eur-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","Bińkowski and Wojdacz Genome Medicine (2026) 18:66  \n[https://doi.org/10.1186/s13073-026-01650-w](https://doi.org/10.1186/s13073-026-01650-w)  \nGenome Medicine  \nRESEARCH Open Access  \nDNA methylation biomarkers-based pan- cancer classifier: predictive modeling  \nfor cancer classification  \nJan Bińkowski 1,2 and Tomasz K. Wojdacz 1,2*  \nAbstract  \nBackground Machine-learning (ML) driven molecular diagnostics based on omics data has a potential to revolutionize personalized medicine. However, implementation of ML into diagnostic protocols is hindered by methodological challenges which often lead to inflated performance assessment of models during development followed by poor performance of these models in implementation phase. Here, we aimed to develop and validate a pan-cancer classification framework based on DNA methylation data, that addresses methodological challenges of omics data powered ML.  \nMethods We curated a primary dataset of DNA methylation profiles for 10756 samples, that included 54 healthy and cancer tissue types and validation dataset comprising data for 2306 samples from 28 independent studies. The classification framework was build using custom biomarkers selection strategy based on effect size metric that considers variance and class imbalance. The ML models were trained, tuned and evaluated using nested crossvalidation approach. Local outlier factor algorithm was built into the inference pipelines to identify and filter samples displaying technical or biological anomalies. Additionally, for methodological validation of our framework we used methylation profiles for 3905 central nervous system (CNS) tumors.  \nResults We found that relatively simple ML models outperformed complex algorithms such as deep neural network. A logistic regression classifier achieved a balanced accuracy (BACC) of 0.90 to classify 54 cancer and healthy tissue types using methylation levels at 1208 CpG sites. Similarly, our CNS tumor classifier also based on logistic regression algorithm reached a BACC of 0.94 across 59 CNS tumor subtypes. The anomaly filtering improved performance across all categories of samples tested.  \nConclusions Our study demonstrates that DNA methylation profiling, when combined with carefully controlled ML practices allows for development of robust solutions that might substantially increase the efficacy of oncological diagnosis. Finally, we deployed our inference pipelines for public access via secure web platform-[https://opp.pum](https://opp.pum). [edu.pl/](edu.pl/) .  \nKeywords DNA methylation, Machine-learning, Biomarkers, Cancer, Classification  \n*Correspondence:  \nTomasz K. Wojdacz  \n[tomasz.wojdacz@pum.edu.pl](tomasz.wojdacz@pum.edu.pl)  \n1Independent Clinical Epigenetics Laboratory, Pomeranian Medical University in Szczecin, Szczecin, Poland  \n2Regional Center for Digital Medicine, Pomeranian Medical University in Szczecin, Aleja Powstańców Wielkopolskich 72, Szczecin 71-899, Poland  \n© The Author(s) 2026. Open Access This article is licensed under a Creative Commons Attribution-NonCommercial-NoDerivatives 4.0 International License, which permits any non-commercial use, sharing, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if you modified the licensed material. You do not have permission under this licence to share adapted material derived from this article or parts of it. The images or other third party material in this article are included in the article’s Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article’s Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit [http://creati](http://creati)[vecommon","cbCailqUTvk58tx8","https://ap.wps.com/l/cbCailqUTvk58tx8","pdf",2501239,"English","# Abstract\n# Background\n## Omics data and diagnostic challenges\n## Methodological obstacles for ML/DL in omics analysis","[{\"question\":\"What problem does the study address in machine-learning molecular diagnostics?\",\"answer\":\"It targets methodological challenges that often cause inflated model performance during development and poor performance when models are implemented in real-world diagnostic settings.\"},{\"question\":\"How is the pan-cancer classification framework developed and validated?\",\"answer\":\"The study curates a primary DNA methylation dataset and a validation dataset from independent studies, builds a custom biomarker selection strategy using an effect-size metric, and evaluates models using nested cross-validation.\"},{\"question\":\"Why does the study include anomaly filtering in the inference pipeline?\",\"answer\":\"Local outlier factor is used to identify and filter samples with technical or biological anomalies, which improves classification performance across tested sample categories.\"}]","DNA methylation biomarkers-based pan-cancer classifier - predictive modeling for cancer classification | PDF",1790082133,48]