[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-349622-105":59,"doc-detail-349622-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","hierarchical-modeling-of-tumor-subtypes-in-cell-lines-using-large-scale-genomic-datasets","Hierarchical modeling of tumor subtypes in cell lines using large-scale genomic datasets","","Cancer cell lines are widely used to study tumor biology and predict drug response, yet inaccurate subtype annotations limit translational relevance. A hierarchical classification framework is introduced to align cell line models with patient tumors across resolutions, from organ to molecular subtype. By integrating gene expression profiles from 802 cell lines, 5,612 TCGA tumors, and 8,939 non-cancerous tissues, the method separates oncogenic from tissue-specific signals using node-specific feature selection. It reassigns 43 cell lines and identifies clinically relevant underrepresented subtypes.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/hierarchical-modeling-of-tumor-subtypes-in-cell-lines-using-large-scale-genomic-datasets/349622/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/hierarchical-modeling-of-tumor-subtypes-in-cell-lines-using-large-scale-genomic-datasets/349622.png","ImageObject",300,407,{"name":92,"@type":93},"CatatanPagi","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-23","2026-09-22",true,{"@type":102,"interactionType":103,"userInteractionCount":8},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"Why are existing cancer cell line subtype annotations considered unreliable?","Question",{"text":112,"@type":113},"Cell lines may diverge from the original tumors during derivation and passaging due to genetic drift, clonal selection, or overgrowth. Sample mislabeling and contamination can also cause incorrect tumor identity, making nominal annotations insufficient.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"What hierarchical classification framework is proposed to improve cell line–tumor matching?",{"text":117,"@type":113},"The framework assigns cell lines to patient tumors across biological resolutions, from organ to molecular subtype. It integrates gene expression profiles to distinguish oncogenic signals from tissue-specific signals.",{"name":119,"@type":110,"acceptedAnswer":120},"How is performance evaluated, and what key outcomes are reported?",{"text":121,"@type":113},"Balanced accuracies of 89% in cross-validation and 75% and 80% on external datasets are reported. The framework also reassigns 43 cell lines and highlights clinically relevant underrepresented tumor subtypes.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},349622,1790197191,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":8,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":144,"read_time":36},962090894170,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","iScience Article  \nHierarchical modeling of tumor subtypes in cell lines using large-scale genomic datasets  \nAuthors  \nJuho Mikkonen, Teemu J. Rintala, Vittorio Fortino  \nCorrespondence  \n[vittorio.fortino@uef.fi](vittorio.fortino@uef.fi)  \nIn brief  \nBioinformatics; Cancer; Machine learning  \n• Hierarchical models assigned tissue, cancer type, and subtype identities  \n• Several cell lines were not representative of any related tumor subtypes  \n• Many tumor subtypes remain poorly represented by existing cell line panels  \nMikkonen et al., 2026, iScience 29, 117315  \nSeptember 18, 2026 © 2026 Published by Elsevier Inc.  \n[https://doi.org/10.1016/j.isci.2026.117315](https://doi.org/10.1016/j.isci.2026.117315)  \nll  \niScience  \nll  \nOPEN ACCESS  \nArticle  \nHierarchical modeling of tumor subtypes in cell lines using large-scale genomic datasets  \nJuho Mikkonen,1 Teemu J. Rintala,1 and Vittorio Fortino1,2,3,*  \n1Institute of Biomedicine, University of Eastern Finland, 70210 Kuopio, Finland  \n2Senior author  \n3Lead contact  \n*[Correspondence:](Correspondence: vittorio.fortino@uef.fi)[ vittorio.fortino@uef.fi](Correspondence: vittorio.fortino@uef.fi)[ ](Correspondence: vittorio.fortino@uef.fi)[https://doi.org/10.1016/j.isci.2026.117315](https://doi.org/10.1016/j.isci.2026.117315)  \nSUMMARY  \nCancer cell lines (CLs) are widely used to study tumor biology and drug response, yet their translational relevance is often limited by inaccurate subtype annotations. Existing CL-tumor matching approaches are frequently constrained by flat classification schemes, weak subtype definitions, and the exclusion of normal tissue references, leading to potential confounding of tumor-specific and tissue-of-origin signals. To address these limitations, a hierarchical classification (HC) framework is presented in which CLs are aligned with patient tumors across biological resolutions, from organ to molecular subtype. Gene expression profiles from 802 CLs, 5,612 tumors from The Cancer Genome Atlas (TCGA) , and 8,939 non-cancerous tissues were integrated to separate oncogenic signals from tissue-specific signals. Node-specific features were selected using maximum relevance minimum redundancy, and balanced accuracies of 89% in cross-validation and 75%, and 80% on external datasets were achieved. Through the framework, 43 CLs were reassigned, and clinically relevant underrepresented subtypes were identified.  \nINTRODUCTION  \nCell lines (CLs) are efficient tools for studying biological mechanisms, drug responses, and gene function while avoiding the complexities of primary tissues and whole organisms. They are cost-effective, consistent, readily scalable, and amenable to high-throughput sequencing. As a result, large resources such as the Cancer Cell Line Encyclopedia (CCLE) and the Genomics of Drug Sensitivity in Cancer (GDSC) link multi-omics readouts with responses to anticancer agents across diverse cancer types and subtypes, offering opportunities for precision medicine by informing therapy choices from molecular profiles.1 Despite this promise, the clinical translatability of CL resources remains uncertain. CLs are typically used under the assumption that they preserve the genomic and epigenomic features of the tumors they model. In practice, they may diverge during derivation and passaging through genetic drift, clonal selection, or overgrowth of unrepresentative populations.2 Additionally, original sample mislabeling due to clinician errors, contamination, or sampling from metastatic lesions has been reported.3 As a result, the nominal annotations of CLs cannot be assumed to reliably reflect their true tumor identity, creating the need to develop algorithms that address this technical gap by improving the alignment between in vitro cancer CL models and available cancer patient data. Such misalignment may have important consequences, particularly for the use of high-throughput drug screening data generated in vitro. Therapeutic vulnerabilities identifi","cbCaisBheQGG4qRN","https://ap.wps.com/l/cbCaisBheQGG4qRN","pdf",4026533,16,"English","# Summary\n## Methods and data integration\n## Results and subtype reassignment\n# Introduction\n## Need for accurate cell line–tumor alignment\n## Prior approaches and limitations\n## Building on existing work","[{\"question\":\"Why are existing cancer cell line subtype annotations considered unreliable?\",\"answer\":\"Cell lines may diverge from the original tumors during derivation and passaging due to genetic drift, clonal selection, or overgrowth. Sample mislabeling and contamination can also cause incorrect tumor identity, making nominal annotations insufficient.\"},{\"question\":\"What hierarchical classification framework is proposed to improve cell line–tumor matching?\",\"answer\":\"The framework assigns cell lines to patient tumors across biological resolutions, from organ to molecular subtype. It integrates gene expression profiles to distinguish oncogenic signals from tissue-specific signals.\"},{\"question\":\"How is performance evaluated, and what key outcomes are reported?\",\"answer\":\"Balanced accuracies of 89% in cross-validation and 75% and 80% on external datasets are reported. The framework also reassigns 43 cell lines and highlights clinically relevant underrepresented tumor subtypes.\"}]","Hierarchical modeling of tumor subtypes in cell lines using large-scale genomic datasets | PDF",1790084563]