[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-141011-105":3,"detail-sidebar-cat-0-en-105":80,"doc-detail-141011-en":130},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":73,"head_meta":75,"extra_data":77,"updated_unix":79},105,"en","zeus-zero-shot-embeddings-for-unsupervised-separation-of-tabular-data","ZEUS: Zero-shot Embeddings for Unsupervised Separation of Tabular Data","","Clustering tabular data remains a central challenge in machine learning because similarity relationships vary strongly across datasets and meaningful cluster definitions are highly dataset-dependent. The lack of supervised signals further makes hyperparameter tuning in deep clustering unstable. ZEUS addresses these issues using zero-shot learning: it clusters new tabular datasets without extra training or fine-tuning by decomposing datasets into meaningful components and leveraging pretraining on synthetic data.",{"@graph":14,"@context":72},[15,34,55],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/research-report/","Research & Report",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/zeus-zero-shot-embeddings-for-unsupervised-separation-of-tabular-data/141011/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/zeus-zero-shot-embeddings-for-unsupervised-separation-of-tabular-data/141011.png","ImageObject",300,407,{"name":42,"@type":43},"Chloe Bennett","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-11","2026-08-25",true,{"@type":52,"interactionType":53,"userInteractionCount":26},"InteractionCounter",{"@type":54},"ViewAction",{"@type":56,"mainEntity":57},"FAQPage",[58,64,68],{"name":59,"@type":60,"acceptedAnswer":61},"Why is clustering tabular data especially difficult compared with images or texts?","Question",{"text":62,"@type":63},"Tabular records lack consistent spatial or semantic properties, and similarity between rows is highly dataset-specific, making generalization across applications hard.","Answer",{"name":65,"@type":60,"acceptedAnswer":66},"How does ZEUS perform clustering without supervised signals or fine-tuning?",{"text":67,"@type":63},"ZEUS is a zero-shot model that returns an embedding for a new dataset in a representation suitable for clustering, using a frozen transformer and a single forward pass.",{"name":69,"@type":60,"acceptedAnswer":70},"What enables ZEUS to generalize across different tabular datasets?",{"text":71,"@type":63},"It is pretrained on synthetic datasets generated from a latent-variable prior, learning to infer cluster assignments under diverse prior clustering structures.","https://schema.org",{"og:url":32,"og:type":74,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":76,"canonical":32},"index,follow",{"doc_id":78,"site_id":7},141011,1787647716,{"code":4,"msg":81,"data":82},"success",[83,87,91,95,100,105,110,114,119,122,126],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":84,"show_sort_weight":85,"slug":86},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":88,"show_sort_weight":89,"slug":90},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":92,"show_sort_weight":93,"slug":94},"Exam",70,"exam",{"id":96,"doc_module":4,"doc_module_name":25,"category_name":97,"show_sort_weight":98,"slug":99},5,"Comic",60,"comic",{"id":101,"doc_module":4,"doc_module_name":25,"category_name":102,"show_sort_weight":103,"slug":104},6,"Technology",50,"technology",{"id":106,"doc_module":4,"doc_module_name":25,"category_name":107,"show_sort_weight":108,"slug":109},7,"Healthcare",40,"healthcare",{"id":111,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":112,"slug":113},8,30,"research-report",{"id":115,"doc_module":4,"doc_module_name":25,"category_name":116,"show_sort_weight":117,"slug":118},9,"Religion & Spirituality",20,"religion-spirituality",{"id":117,"doc_module":4,"doc_module_name":25,"category_name":120,"show_sort_weight":117,"slug":121},"World Cup","world-cup",{"id":123,"doc_module":4,"doc_module_name":25,"category_name":124,"show_sort_weight":123,"slug":125},10,"Lifestyle","lifestyle",{"id":127,"doc_module":4,"doc_module_name":25,"category_name":128,"show_sort_weight":96,"slug":129},19,"General","general",{"code":4,"msg":81,"data":131},{"doc_id":78,"user_id":132,"nickname":42,"user_avatar":133,"doc_module":4,"category_id":111,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":26,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":112,"language":139,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":12,"update_tm":79,"read_time":143},962084925782,"https://ap-avatar.wpscdn.com/davatar_9964176cb1d06d4a9deccf72a44ae3dc","arXiv :2505 . 10704v2 [ cs .LG] 24 Oct 2025  \nZEUS: Zero-shot Embeddings for Unsupervised Separation of Tabular Data  \nPatryk Marszałek 1 ,2 Tomasz Kumierczyk∗ , 1 Witold Wydmaski 1 ,2 Jacek Tabor 1  \nMarek ´Smieja∗ , 1  \n1Faculty of Mathematics and Computer Science, Jagiellonian University, Kraków, Poland  \n2Doctoral School of Exact and Natural Sciences, Jagiellonian University, Kraków, Poland  \n{t.kusmierczyk,witold.wydmanski,jacek.tabor,[marek.smieja}@uj.edu.pl](marek.smieja}@uj.edu.pl)[ ](marek.smieja}@uj.edu.pl)[patryk.marszalek@doctoral.uj.edu.pl](patryk.marszalek@doctoral.uj.edu.pl)  \nAbstract  \nClustering tabular data remains a significant open challenge in data analysis and machine learning. Unlike for image data, similarity between tabular records often varies across datasets, making the definition of clusters highly dataset-dependent.  \nFurthermore, the absence of supervised signals complicates hyperparameter tuning in deep learning clustering methods, frequently resulting in unstable performance.  \nTo address these issues and reduce the need for per-dataset tuning, we adopt an emerging approach in deep learning: zero-shot learning. We propose ZEUS, a selfcontained model capable of clustering new datasets without any additional training or fine-tuning. It operates by decomposing complex datasets into meaningful components that can then be clustered effectively. Thanks to pre-training on synthetic datasets generated from a latent-variable prior, it generalizes across various datasets without requiring user intervention. To the best of our knowledge, ZEUS is the first zero-shot method capable of generating embeddings for tabular data in a fully unsupervised manner. Experimental results demonstrate that it performs on par with or better than traditional clustering algorithms and recent deep learning-based methods, while being significantly faster and more user-friendly.  \n1 Introduction  \nClustering remains a fundamental yet challenging task in unsupervised learning. It is particularly hard for tabular data, which inherently lacks the structured spatial or semantic properties of images or texts. Unlike for image clustering, where intrinsic visual similarities can guide cluster formation, defining meaningful similarities in tabular data is highly dataset-specific, complicating the generalization of clustering methods across diverse applications. Recent developments leveraging deep learning have demonstrated promise in generating richer representations for clustering tasks. However, these methods frequently suffer from instability due to their sensitivity to hyperparameter selection, a challenge exacerbated by the absence of supervised signals to guide optimization. Consequently, practitioners working with tabular data often resort to simpler, classical algorithms like k-means [23], despite their limited capacity for capturing complex underlying data structures, simply to avoid extensive manual tuning.  \n∗Joint contribution to project conception, design, and research supervision.  \nAccepted to the 39th Conference on Neural Information Processing Systems (NeurIPS 2025) .  \nFigure 1: Schematic characterization of ZEUS: (left) synthetic datasets generation; (middle) pretraining on datasets with known labels; (right) deployment of a frozen model for real-world tasks.  \nWe address these challenges with ZEUS – a zero-shot transformer-based model for embedding new tabular datasets in a form convenient for unsupervised separation (=clustering), without the need for additional fine-tuning. Given a new dataset, ZEUS returns its transformed representation, where clusters can be easily discovered using simple methods, like k-means. Since it works as a zero-shot learner, it significantly reduces hyperparameter tuning complexity and computation time, enabling effective clustering in seconds. It is a plug-and-play solution that requires no fine-tuning and runs ina single forward pass for new datasets.  \nInspiration for ZEUS stems from ","cbCaifBjgxv6MiFq","https://ap.wps.com/l/cbCaifBjgxv6MiFq","pdf",1899430,"English","# Abstract\n# 1 Introduction\n## Problem: dataset-dependent similarities and unstable tuning\n## Solution: ZEUS zero-shot transformer embeddings\n## Inspiration and method overview","[{\"question\":\"Why is clustering tabular data especially difficult compared with images or texts?\",\"answer\":\"Tabular records lack consistent spatial or semantic properties, and similarity between rows is highly dataset-specific, making generalization across applications hard.\"},{\"question\":\"How does ZEUS perform clustering without supervised signals or fine-tuning?\",\"answer\":\"ZEUS is a zero-shot model that returns an embedding for a new dataset in a representation suitable for clustering, using a frozen transformer and a single forward pass.\"},{\"question\":\"What enables ZEUS to generalize across different tabular datasets?\",\"answer\":\"It is pretrained on synthetic datasets generated from a latent-variable prior, learning to infer cluster assignments under diverse prior clustering structures.\"}]","ZEUS: Zero-shot Embeddings for Unsupervised Separation of Tabular Data | PDF",76]