[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-141104-105":59,"doc-detail-141104-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","do-large-language-models-know-folktales-a-case-study-of-yokai-in-japanese-folktales","Do Large Language Models Know Folktales - A Case Study of Yokai in Japanese Folktales","","Large Language Models (LLMs) demonstrate strong multilingual language understanding and generation, yet their cultural knowledge can be restricted to English-speaking communities. To reduce this gap, the study evaluates cultural awareness through folktale knowledge, with a focus on Japanese yokai as enduring cultural motifs across art and entertainment. It introduces YokaiEval, a benchmark of 809 multiple-choice questions covering yokai facts. Experiments on 31 Japanese and multilingual LLMs show higher accuracy for models trained with Japanese resources, especially those that continued pretraining in Japanese based on Llama-3.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/do-large-language-models-know-folktales-a-case-study-of-yokai-in-japanese-folktales/141104/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/do-large-language-models-know-folktales-a-case-study-of-yokai-in-japanese-folktales/141104.png","ImageObject",300,407,{"name":92,"@type":93},"Mia  ","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-15","2026-08-25",true,{"@type":102,"interactionType":103,"userInteractionCount":24},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What cultural ability does the study evaluate in LLMs?","Question",{"text":112,"@type":113},"The study evaluates LLMs’ knowledge of folktales as a proxy for cultural awareness, specifically Japanese yokai knowledge.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"What is YokaiEval?",{"text":117,"@type":113},"YokaiEval is a benchmark dataset of 809 multiple-choice questions designed to probe factual knowledge about yokai.",{"name":119,"@type":110,"acceptedAnswer":120},"Which LLMs perform better on YokaiEval and why?",{"text":121,"@type":113},"Japanese-resource-trained models achieve higher accuracy, particularly models that continued pretraining in Japanese, especially those based on Llama-3.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},141104,1787649101,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":24,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},687207024478,"https://ap-avatar.wpscdn.com/davatar_a8503ba1806abce46bf441b54a3ca4cd","Do Large Language Models Know Folktales? A Case Study of Yokai in Japanese Folktales  \nTsutsumi Ayuto  \nTokyo Metropolitan University [tsutsumi-ayuto@ed.tmu.ac.jp](tsutsumi-ayuto@ed.tmu.ac.jp)  \nYuu Jinnai  \nCyberAgent [jinnai_yu@cyberagent.co.jp](jinnai_yu@cyberagent.co.jp)  \nAbstract  \nAlthough Large Language Models (LLMs) have demonstrated strong language understanding and generation abilities across various languages, their cultural knowledge is often limited to English-speaking communities, which can marginalize the cultures of non-English communities. To address the problem, evaluation of the cultural awareness of the LLMsand the methods to develop culturally aware LLMs have been investigated. In this study, we focus on evaluating knowledge of folktales, a key medium for conveying and circulating culture. In particular, we focus on Japanese folktales, specifically on knowledge of Yokai. Yokai are supernatural creatures originating from Japanese folktales that continue to be popular motifs in art and entertainment today. Yokai have long served as a medium for cultural expression, making them an ideal subject for assessing the cultural awareness of LLMs. We introduce YokaiEval, a benchmark dataset consisting of 809 multiple-choice questions (each with four options) designed to probe knowledge about yokai. We evaluate the performance of 31 Japanese and multilingual LLMs on this dataset. The results show that models trained with Japanese language resources achieve higher accuracy than English-centric models, with those that underwent continued pretraining in Japanese, particularly those based on Llama-3, performing especially well. The code and dataset are available at [https://github.com/CyberAgentA](https://github.com/CyberAgentA)ILab/YokaiEval.  \n1 Introduction  \nLarge Language Models (LLM) have shown remarkable performance in language understanding and generation tasks (Ouyang et al., 2022 ; Touvron et al., 2023 ; OpenAI et al., 2024) . Despite many LLMs being predominantly trained in English, their generalization capabilities allow them to transfer knowledge across languages, achieving  \nFigure 1: The Night Parade of One Hundred Demons (百鬼夜行) by Kawanabe Kyosai. It is said that a parade of supernatural creatures known as yokai march through the streets of Japan at night, and anyone who comes across would be spirited away. The folktale became one of the popular motifs in the Edo period portrayed in many media including ukiyo-e, toys, and picture scrolls.  \ndecent performance even in resource-limited languages (Chen et al., 2023 ; Shaham et al., 2024 ; OpenAI et al., 2024) .  \nWhile there is evidence for strong cross-lingual transfer capability, cross-cultural transfer is known to be challenging for LLMs (Arango Monnar et al., 2022 ; Hershcovich et al., 2022 ; Lee et al., 2023 ; Huang and Yang, 2023 ; Rao et al., 2024 ; Adilazuarda et al., 2024 ; Cao et al., 2024 ; Liu et al., 2024) . Prior work shows that LLMs tend to be biased toward the values and opinions of certain communities, rather than representing the diversity of human values (Santurkar et al., 2023 ; Conitzer et al., 2024) .  \nTo address this issue, many studies have investigated methods to evaluate the cultural awareness of LLMs (Rao et al., 2024 ; Adilazuarda et al., 2024 ; Liu et al., 2024) . Simultaneously, approaches to develop culturally aware LLMs using the language resources of target communities are being explored (Pires et al., 2023 ; Lin and Chen, 2023 ; Nguyen et al., 2023 ; Huang et al., 2024 ; Owen et al., 2024 ;  \n16124  \nFindings of the Association for Computational Linguistics: ACL 2025 , pages 16124–16146 July 27-August 1, 2025 ©2025 Association for Computational Linguistics  \nTran et al., 2024 ; Etxaniz et al., 2024) .  \nWhile many prior studies have investigated differences in the values and opinions of communities (Xu et al., 2024 ; Sorensen et al., 2024 ; Wang et al., 2024a ; Naous et al., 2024 ; Durmus et al., 2024), the medium that conv","cbCaitVsv7zUnxKm","https://ap.wps.com/l/cbCaitVsv7zUnxKm","pdf",1905538,23,"English","# Introduction\n# Yokai: From Traditional Folktales to Today’s Art and Entertainment","[{\"question\":\"What cultural ability does the study evaluate in LLMs?\",\"answer\":\"The study evaluates LLMs’ knowledge of folktales as a proxy for cultural awareness, specifically Japanese yokai knowledge.\"},{\"question\":\"What is YokaiEval?\",\"answer\":\"YokaiEval is a benchmark dataset of 809 multiple-choice questions designed to probe factual knowledge about yokai.\"},{\"question\":\"Which LLMs perform better on YokaiEval and why?\",\"answer\":\"Japanese-resource-trained models achieve higher accuracy, particularly models that continued pretraining in Japanese, especially those based on Llama-3.\"}]","Do Large Language Models Know Folktales - A Case Study of Yokai in Japanese Folktales | PDF",58]