[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-349623-105":59,"doc-detail-349623-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","lung-cancer-risk-prediction-using-interpretable-ensemble-models-on-lifestyle-and-clinical-data","Lung cancer risk prediction using interpretable ensemble models on lifestyle and clinical data","","Lung cancer remains a leading cause of cancer-related mortality, making early detection and reliable risk assessment essential. The study develops and evaluates interpretable ensemble learning models that use lifestyle and clinical indicators, prioritizing both predictive accuracy and interpretability. Five base learners—logistic regression, k-nearest neighbors, naïve Bayes, support vector machine, and linear discriminant analysis—feed boosting and bagging variants. Voting and stacking ensembles are built by selectively combining high-performing models, assessed on original, balanced, and upsampled datasets using accuracy, precision, recall, F1-score, MCC, and AUC-ROC. Stacking achieves the strongest overall performance, and SHAP plus LIME enable global and local explanations to identify key factors and patient-specific risk drivers.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/lung-cancer-risk-prediction-using-interpretable-ensemble-models-on-lifestyle-and-clinical-data/349623/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/lung-cancer-risk-prediction-using-interpretable-ensemble-models-on-lifestyle-and-clinical-data/349623.png","ImageObject",300,407,{"name":92,"@type":93},"CatatanPagi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-26","2026-09-22",true,{"@type":102,"interactionType":103,"userInteractionCount":19},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What modeling approach is used for lung cancer risk prediction?","Question",{"text":112,"@type":113},"The study uses ensemble learning with five base learners to build boosting, bagging, voting, and stacking models for prediction.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How is interpretability achieved in the models?",{"text":117,"@type":113},"SHAP is used for global interpretability and LIME for local interpretability to explain key clinical factors and patient-specific risk drivers.",{"name":119,"@type":110,"acceptedAnswer":120},"How are model performances evaluated in the experiments?",{"text":121,"@type":113},"Performance is measured on original, balanced, and upsampled datasets using accuracy, precision, recall, F1-score, MCC, and AUC-ROC, with stacking delivering the best overall results.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},349623,1790139746,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":19,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":145},962090894170,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","OPEN ACCESS  \nCitation: Ganie SM, Dutta Pramanik PK, Zhao Z (2026) Lung cancer risk prediction using interpretable ensemble models on lifestyle and clinical data. PLoS One 21(9): e0357291 .  \n[https://doi.org/10.1371/journal.pone.0357291](https://doi.org/10.1371/journal.pone.0357291)  \n[Editor:](Editor: Amgad Muneer)[ Amgad Muneer](Editor: Amgad Muneer), [The University of Texas](The University of Texas), MD Anderson Cancer Center, UNITED STATES OF AMERICA  \nReceived: September 22, 2025  \nAccepted: August 13, 2026  \nPublished: September 3, 2026  \nPeer Review History: PLOS recognizes the benefits of transparency in the peer review process; therefore, we enable the publication of all of the content of peer review and author responses alongside final, published articles. The editorial history of this article is available here: [https://doi.org/10.1371/journal](https://doi.org/10.1371/journal). pone.0357291  \nCopyright: This is an open access article, free of all copyright, and may be freely reproduced, distributed, transmitted, modified, built upon, or otherwise used by anyone for any lawful purpose. The work is made available under  \nRESEARCH ARTICLE  \nLung cancer risk prediction using interpretable ensemble models on lifestyle and clinical data  \nShahid Mohammad Ganie1, Pijush Kanti Dutta Pramanik2,3*, Zhongming Zhao3*  \n1 Department of Health Information Management and Technology, College of Applied Medical Sciences, King Faisal University, Al-Ahsa, Saudi Arabia, 2 School of Computer Applications and Technology, Galgotias University, Greater Noida, Uttar Pradesh, India, 3 Center for Precision Health, McWilliams School of Biomedical Informatics, The University of Texas Health Science Center at Houston, Houston, Texas, United States of America  \n* [pijushjld@yahoo.co.in](pijushjld@yahoo.co.in) (PKDP); [zhongming.zhao@uth.tmc.edu](zhongming.zhao@uth.tmc.edu) (ZZ)  \nAbstract  \nLung cancer remains one of the leading causes of cancer-related mortality worldwide, where early detection and reliable risk assessment are critical for improving outcomes. This study develops and evaluates a range of ensemble learning models for lung cancer prediction using lifestyle and clinical indicators, with an emphasis on both predictive performance and interpretability. Five base learners—logistic regression, k-nearest neighbors, naïve Bayes, support vector machine, and linear discriminant analysis—were used to construct multiple boosting and bagging models. Building on these, voting and stacking ensembles were designed by selectively combining high-performing models. All approaches were evaluated on the original dataset as well as on balanced and upsampled variants derived through synthetic augmentation. Model performance was assessed using accuracy, precision, recall, F1-score, Matthews correlation coefficient (MCC), and AUC-ROC. The results show that ensemble approaches consistently outperform individual models, with voting and stacking demonstrating superior performance over boosting and bagging methods. The stacking model achieved the strongest overall performance across all evaluated models. On the original dataset, which provides a more realistic representation of practical deployment conditions, it attained an accuracy of 93.53% . Performance further improved on the balanced and upsampled datasets on the upsampled dataset. To enhance transparency, SHAP and LIME were employed to provide global and local interpretability, respectively, enabling identification of key clinical factors and patient-specific risk drivers. The analysis highlights both alignment with known clinical patterns and dataset-driven variations, supporting informed interpretation of model outputs. The results suggest that stacking-based ensembles can improve risk prediction from lifestyle and clinical indicators while maintaining model transparency through SHAP and LIME explanations. These findings highlight the potential  \nPLOS One | [https://doi.org/10.1371/journal.pone.035","cbCaisREExY4LZHZ","https://ap.wps.com/l/cbCaisREExY4LZHZ","pdf",5835486,45,"English","# Abstract\n# Introduction\n## Motivation for early detection\n## Distinguishing benign and malignant cases","[{\"question\":\"What modeling approach is used for lung cancer risk prediction?\",\"answer\":\"The study uses ensemble learning with five base learners to build boosting, bagging, voting, and stacking models for prediction.\"},{\"question\":\"How is interpretability achieved in the models?\",\"answer\":\"SHAP is used for global interpretability and LIME for local interpretability to explain key clinical factors and patient-specific risk drivers.\"},{\"question\":\"How are model performances evaluated in the experiments?\",\"answer\":\"Performance is measured on original, balanced, and upsampled datasets using accuracy, precision, recall, F1-score, MCC, and AUC-ROC, with stacking delivering the best overall results.\"}]","Lung cancer risk prediction using interpretable ensemble models on lifestyle and clinical data | PDF",1790084564,113]