[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-304613-105":53,"doc-detail-304613-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","surprisal-and-crossword-clues-difficulty-evaluating-linguistic-processing-between-llms-and-humans","Surprisal and Crossword Clues Difficulty - Evaluating Linguistic Processing between LLMs and Humans","","Crossword clue difficulty is traditionally determined by human setters, leaving automated generators without an objective yardstick. The study models difficulty as the Surprisal of the intended answer given a clue, estimating Surprisal using token probabilities from large language models. It compares three causal LLMs—Llama-3-8B, Llama-2-7B, and Ita-GPT-2-121M—with 60 human solvers on 160 hand-balanced clues, finding negative correlation between Surprisal and accuracy.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":11,"@type":70,"position":76},"https://docshare.wps.com/template/presentations/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/surprisal-and-crossword-clues-difficulty-evaluating-linguistic-processing-between-llms-and-humans/304613/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/surprisal-and-crossword-clues-difficulty-evaluating-linguistic-processing-between-llms-and-humans/304613.png","ImageObject",442,249,{"name":88,"@type":89},"Aladdin","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-28","2026-09-19",true,{"@type":98,"interactionType":99,"userInteractionCount":76},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"How is crossword clue difficulty measured in this paper?","Question",{"text":108,"@type":109},"Difficulty is modeled as the Surprisal of the answer given the clue, computed from token probabilities produced by large language models.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"Which language models are compared, and how do results relate to human solving?",{"text":113,"@type":109},"The study compares Llama-3-8B, Llama-2-7B, and Ita-GPT-2-121M against 60 human solvers on 160 balanced clues, showing that higher Surprisal corresponds to lower accuracy.",{"name":115,"@type":106,"acceptedAnswer":116},"What are the proposed applications of the Surprisal metric for crossword puzzles?",{"text":117,"@type":109},"The metric can support adaptive crossword generation, tutoring tools, fairness in online competitions, and psycholinguistic experimentation focused on alignment between human and model linguistic processing.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},304613,1790488346,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":8,"category_name":11,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":76,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":20,"language":135,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":136,"faqs":137,"seo_title":138,"seo_description":61,"update_tm":139,"read_time":47},2336478503145,"https://ap-avatar.wpscdn.com/davatar_276721f389ce27ea32af1340a28f341c","Surprisal and Crossword Clues difficulty: Evaluating Linguistic Processing between LLMs and Humans  \nTommaso Iaquinta1, *,†, Asya Zanollo2,3,†, Achille Fusco3,4,†, Kamyar Zeinalipour1,† and Cristiano Chesi2,3,†  \n1 Università degli Studi di Siena (UNISI), Via Roma 56, 53100 Siena, Italy  \n2 University School for Advanced Studies IUSS Pavia, Piazza della Vittoria 15, 27100 Pavia, Italy  \n3 Laboratory for Neurocognition, Epistemology, and Theoretical Syntax -NeTS-IUSS Pavia  \n4 Università degli Studi di Firenze, Piazza S. Marco 4, 50121 Firenze, Italy  \nAbstract  \nCrossword clue difficulty is traditionally judged by human setters, leaving automated puzzle generators without an objective yard-stick. We model difficulty as the Surprisal of the answer given the clue, estimating it with token probabilities from large language models. Comparing three models three causal LLMs-Llama-3-8B, Llama-2-7B, and Ita-GPT-2-121M. with 60 human solvers on 160 hand-balanced clues, Surprisal correlates negatively with accuracy (r = –0.62 for nominal clues) . These results show that language-model Surprisal captures some of the cognitive load humans experience and that language-specific training and model scale both matter; the metric therefore enables adaptive crossword generation and provides a new test-bed for probing the alignment between human and model linguistic processing.  \nKeywords  \nsurprisal, llm, gpt, crossword, education, linguistic games, puzzle, Crossword difficulty  \n1. Introduction  \nCrossword (CW) puzzles are among the most popular language games, captivating millions through newspapers, mobile apps, voice assistants, and even televised competitions [1, 2] . The enduring appeal of crosswords across formats stems from the careful calibration of clue difficulty, which can range from accessible, beginner-friendly  \nprompts to highly intricate, expert-level challenges.  \nDespite advancements in automated puzzle generation, state-of-the-art systems like Dr. Fill [3] and the Berkeley Crossword Solver [1], while capable of outperforming many human solvers, still lack a reliable, objective measure to assess the challenge posed by the clues they generate. Traditional heuristics, such as clue length, grid density, historical solve statistics, and letter  \nCLiC-it 2025: Eleventh Italian Conference on Computational Linguistics, September 24 — 26, 2025, Cagliari, Italy  \n* Corresponding author.  \n† † These authors contributed equally.  \n$ [tommaso.iaquinta@unisi.it](tommaso.iaquinta@unisi.it) (T. Iaquinta); [asya.zanollo@iusspavia.it](asya.zanollo@iusspavia.it) (A. Zanollo); [achille.fusco@iusspavia.it](achille.fusco@iusspavia.it)[ ](achille.fusco@iusspavia.it)(A. Fusco); [kamyar.zeinalipour2@unisi.it](kamyar.zeinalipour2@unisi.it) (K. Zeinalipour);  \n[cristiano.chesi@iusspavia.it](cristiano.chesi@iusspavia.it) (C. Chesi)  \n􀂀 https://tommyiaq.me (T. Iaquinta); [https://www.iusspavia.it/it/rubrica/asya-zanollo/](https://www.iusspavia.it/it/rubrica/asya-zanollo/) (A. Zanollo); [https://github.com/achille-fusco](https://github.com/achille-fusco) (A. Fusco); [https://kamyarzeinalipour.github.io/](https://kamyarzeinalipour.github.io/) (K. Zeinalipour); [https://github.com/cristianochesi](https://github.com/cristianochesi) (C. Chesi)  \n􀀚 0009-0009-1262-6768 (T. Iaquinta); 0009-0001-3987-4843 (A. Zanollo); 0000-0002-5389-8884 (A. Fusco); 0009-0006-3014-2511 (K. Zeinalipour); 0000-0003-1935-1348 (C. Chesi)  \n© 2025 Copyright for this paper by its authors. Use permitted under Creative Commons License Attribution 4 .0 International (CC BY 4 .0) .  \nTable 1  \nLinguistic properties for “piante che forniscono frutti per spremute, aranci” (plants that provide fruits for juice – orange trees) .  \n\n| Microcategory | bareNP:rel |\n| --- | --- |\n| Macrocategory | nominal |\n| Accuracy | 0.526 |\n| RTs (log 10) | 4.214 |\n| Surprisal | 5.207 |\n\nTable 2  \nLinguistic properties for “i mobili con le grucce, armadi”(the furniture with hangers – wardrobes) .  \n\n| Microca","cbCaiaALwvmQ7aXq","https://ap.wps.com/l/cbCaiaALwvmQ7aXq","pdf",3280002,"English","# Abstract\n# Introduction","[{\"question\":\"How is crossword clue difficulty measured in this paper?\",\"answer\":\"Difficulty is modeled as the Surprisal of the answer given the clue, computed from token probabilities produced by large language models.\"},{\"question\":\"Which language models are compared, and how do results relate to human solving?\",\"answer\":\"The study compares Llama-3-8B, Llama-2-7B, and Ita-GPT-2-121M against 60 human solvers on 160 balanced clues, showing that higher Surprisal corresponds to lower accuracy.\"},{\"question\":\"What are the proposed applications of the Surprisal metric for crossword puzzles?\",\"answer\":\"The metric can support adaptive crossword generation, tutoring tools, fairness in online competitions, and psycholinguistic experimentation focused on alignment between human and model linguistic processing.\"}]","Surprisal and Crossword Clues Difficulty - Evaluating Linguistic Processing between LLMs and Humans | PDF",1789815429]