[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-327889-105":59,"doc-detail-327889-en":129},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":122,"head_meta":124,"extra_data":126,"updated_unix":128},105,"en","shortcut-learning-in-legal-judgment-prediction-empirical-evidence-from-the-uk-employment-tribunal-327889","Shortcut Learning in Legal Judgment Prediction - Empirical Evidence from the UK Employment Tribunal","","Current Legal Judgment Prediction (LJP) is limited by dependence on post-hoc judicial materials, which increases the chance that models perform retrospective classification rather than true forecasting. This paper empirically examines shortcut learning using claim-level outcome prediction on UK Employment Tribunal decisions with 33,158 claims. Models range from interpretable TF-IDF classifiers to black-box LLMs, and performance rises with leakage cues. Masking leakage features reduces performance only negligibly, showing controllable extraction of predictive signals.",{"@graph":69,"@context":121},[70,84,104],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/shortcut-learning-in-legal-judgment-prediction-empirical-evidence-from-the-uk-employment-tribunal-327889/327889/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":98,"encodingFormat":97,"isAccessibleForFree":99,"interactionStatistic":100},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/shortcut-learning-in-legal-judgment-prediction-empirical-evidence-from-the-uk-employment-tribunal-327889/327889.png","ImageObject",300,407,{"name":92,"@type":93},"Miles","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-21",true,{"@type":101,"interactionType":102,"userInteractionCount":4},"InteractionCounter",{"@type":103},"ViewAction",{"@type":105,"mainEntity":106},"FAQPage",[107,113,117],{"name":108,"@type":109,"acceptedAnswer":110},"Why can Legal Judgment Prediction models fail to perform true forecasting?","Question",{"text":111,"@type":112},"Many LJP datasets rely on post-hoc judicial texts released after the outcome, so models may learn retrospective rationalisations rather than ex ante facts available before a decision.","Answer",{"name":114,"@type":109,"acceptedAnswer":115},"How does the paper test for shortcut learning in UK Employment Tribunal decisions?",{"text":116,"@type":112},"It predicts claim outcomes from claim texts and LLM-extracted case summaries, evaluating both interpretable classifiers and black-box LLMs while stratifying test data by human judgments of leakage.",{"name":118,"@type":109,"acceptedAnswer":119},"What happens when leakage features are masked during retraining?",{"text":120,"@type":112},"Retraining after masking leakage features leads to only a negligible reduction in Macro-F1, indicating models can still extract useful predictive signals without the artefacts.","https://schema.org",{"og:url":83,"og:type":123,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":125,"canonical":83},"index,follow",{"doc_id":127,"site_id":62},327889,1789978274,{"code":4,"msg":5,"data":130},{"doc_id":127,"user_id":131,"nickname":92,"user_avatar":132,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":133,"file_id":134,"file_url":135,"file_type":136,"file_size":137,"view_count":4,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":138,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":128,"read_time":143},13056703019404,"https://ap-avatar.wpscdn.com/davatar_29158cc5080c5b710cf443261637dec0","1  \nShortcut Learning in Legal Judgment Prediction: Empirical Evidence from the UK Employment  \nTribunal  \nJoe Watson1,2*, Joana Ribeiro de Faria1, Marcus Tomalin3, Måns Magnusson4, Huiyuan Xie5, Hao Tian Yeung6, Felix Steffek1  \n1. Faculty of Law, University of Cambridge, Cambridge, United Kingdom  \n2. The Psychometrics Centre, Cambridge Judge Business School, University of Cambridge, Cambridge, United Kingdom  \n3. Faculty of English, University of Cambridge, Cambridge, United Kingdom  \n4. Department of Statistics, Uppsala University, Uppsala, Sweden  \n5. Department of Computer Science and Technology, Tsinghua University, Beijing, China  \n6. Department of Engineering, University of Cambridge, Cambridge, United Kingdom  \n* Corresponding author: Joe Watson, [jmw239@cam.ac.uk](jmw239@cam.ac.uk)  \n2  \nAbstract  \nCurrent Legal Judgment Prediction (LJP) is constrained by its reliance on post-hoc judicial materials, increasing the likelihood that models perform retrospective classification rather than true forecasting. This paper empirically investigates shortcut learning in this context by studying claim-level outcome prediction in UK Employment Tribunal (UKET) decisions. Using a corpus of 33,158 individual claims, we predict outcomes from claim texts and LLMextracted case summaries, evaluating models ranging from interpretable TF-IDF-based classifiers to black-box LLMs. While headline predictive performance figures appear strong, we demonstrate that such performance in LJP systems trained on post-hoc judicial text can be driven by the retrospective nature of the source material. Stratifying the test data by human judgments of leakage reveals that performance increases where outcome-revealing cues are embedded in the narrative. Moreover, a model trained on just the 4% of features identified as leakage achieves high performance, outperforming human experts. These findings substantiate concerns that LJP performance may be exaggerated by linguistic artefacts. Yet this vulnerability is not fatal to the research agenda. Instead, post-hoc judgments might be treated as potentially contaminated texts, requiring active auditing. Retraining models after masking leakage features results in only a negligible reduction in Macro-F1 . Hence, while models will opportunistically exploit shortcuts when available, they remain capable of extracting useful predictive signals when these artefacts are removed.  \nKeywords  \nLegal Judgment Prediction; shortcut learning; information leakage; UK Employment Tribunal  \n3  \nMain Text  \n1 Introduction  \nThe practical promise of Legal Judgment Prediction (LJP) rests on the possibility of forecasting court outcomes from information available before a decision is made, estimating a dispute’s likely resolution from ex ante case materials. In its strongest form, therefore, LJPis a temporally constrained task: a model should not rely on information that would only become available after the relevant legal decision has been reached. This requirement matters because LJP is often motivated by its potential to help litigants, lawyers, or policymakers assess likely outcomes before proceedings. For that promise to hold, predictions must be based on information available at that point, rather than on facts, findings, or reasoning later used to justify the result.  \nThis ideal is difficult to satisfy in practice because most LJP work relies on post-hoc judicial texts: decisions published after the court outcome has been reached. This reliance largely arises from data constraints. Pre-decisional materials, such as claim forms, response forms, comprehensive case files, and other pre-trial documents, are rarely available to researchers (Medvedeva et al. 2023), including in the context of the UK Employment Tribunal (UKET) . As a result, many LJP studies risk performing outcome identification or outcome classification rather than true forecasting (Medvedeva et al. 2023) . The problem is that judicial decisions are not neutral repo","cbCaieaY2v23Dk37","https://ap.wps.com/l/cbCaieaY2v23Dk37","pdf",1180013,25,"English","# Introduction\n## Legal Judgment Prediction as temporally constrained forecasting\n## Reliance on post-hoc judicial text and validity risks\n## Shortcut learning and selective case summaries","[{\"question\":\"Why can Legal Judgment Prediction models fail to perform true forecasting?\",\"answer\":\"Many LJP datasets rely on post-hoc judicial texts released after the outcome, so models may learn retrospective rationalisations rather than ex ante facts available before a decision.\"},{\"question\":\"How does the paper test for shortcut learning in UK Employment Tribunal decisions?\",\"answer\":\"It predicts claim outcomes from claim texts and LLM-extracted case summaries, evaluating both interpretable classifiers and black-box LLMs while stratifying test data by human judgments of leakage.\"},{\"question\":\"What happens when leakage features are masked during retraining?\",\"answer\":\"Retraining after masking leakage features leads to only a negligible reduction in Macro-F1, indicating models can still extract useful predictive signals without the artefacts.\"}]","Shortcut Learning in Legal Judgment Prediction - Empirical Evidence from the UK Employment Tribunal | PDF",63]