[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-267021-105":59,"doc-detail-267021-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","satbench-benchmarking-llms-logical-reasoning-via-automated-puzzle-generation-from-sat-formulas","SATBench - Benchmarking LLMs’ Logical Reasoning via Automated Puzzle Generation from SAT Formulas","","SATBench introduces a benchmark for evaluating large language models’ logical reasoning by automatically generating puzzle instances from Boolean satisfiability (SAT) formulas. Each SAT instance is translated into a natural-language puzzle using LLMs, with difficulty adjusted by varying the number of clauses. All 2100 puzzles are validated through LLM-based and solver-based consistency checks, plus human validation on a subset. Results show limited accuracy on hard UNSAT problems and reveal systematic failures such as satisfiability bias, context inconsistency, and condition omission.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/satbench-benchmarking-llms-logical-reasoning-via-automated-puzzle-generation-from-sat-formulas/267021/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/satbench-benchmarking-llms-logical-reasoning-via-automated-puzzle-generation-from-sat-formulas/267021.png","ImageObject",300,407,{"name":92,"@type":93},"Violet","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-19","2026-09-14",true,{"@type":102,"interactionType":103,"userInteractionCount":14},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is SATBench and what does it benchmark?","Question",{"text":112,"@type":113},"SATBench is a benchmark that evaluates the logical reasoning capabilities of large language models using puzzles generated from Boolean satisfiability (SAT) problems.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How are SATBench puzzles generated and how is difficulty controlled?",{"text":117,"@type":113},"Each puzzle is generated from a sampled SAT formula and translated into a story context with natural-language conditions using LLMs. Difficulty is adjusted by varying the number of clauses in the sampled CNF formulas.",{"name":119,"@type":110,"acceptedAnswer":120},"How are puzzles validated and how is model reasoning evaluated?",{"text":121,"@type":113},"Puzzle quality is verified with both LLM-assisted and solver-based consistency checks, and reasoning traces are assessed using an LLM-as-a-judge strategy in the evaluation pipeline.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},267021,1789413989,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":14,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},4398048950312,"https://ap-avatar.wpscdn.com/avatar/400002538284de19e3c?_k=1778320343897328908","SATBench: Benchmarking LLMs’ Logical Reasoning via Automated Puzzle Generation from SAT Formulas  \nAnjiang Wei1 *†  \nTarun Suresh3  \nSanmi Koyejo 1  \nYuheng Wu1 * Huanmi Tan4 Ke Wang5  \nYingjia Wan2 Zhanke Zhou 1 Alex Aiken 1  \n1 Stanford University 2UCLA 3UIUC 4 CMU 5Nanjing University  \nAbstract  \nWe introduce SATBench, a benchmark for evaluating the logical reasoning capabilities of large language models (LLMs) through logical puzzles derived from Boolean satisfiability (SAT) problems. Unlike prior work that focuses on inference rule-based reasoning, which often involves deducing conclusions from a set of premises, our approach leverages the searchbased nature of SAT problems, where the objective is to find a solution that fulfills a specified set of logical constraints. Each instance in SATBench is generated from a SAT formula, then translated into a puzzle using LLMs. The generation process is fully automated and allows for adjustable difficulty by varying the number of clauses. All 2100 puzzles are validated through both LLM-based and solver-based consistency checks, with human validation on a subset. Experimental results show that even the strongest model, o4-mini, achieves only 65.0% accuracy on hard UNSAT problems, close to the random baseline of 50% . Our error analysis reveals systematic failures such as satisfiability bias, context inconsistency, and condition omission, highlighting limitations of current LLMs in search-based logical reasoning. Our code and data are publicly available at [https:](https:)//[github.com/Anjiang-Wei/SATBench](github.com/Anjiang-Wei/SATBench)  \n1 Introduction  \nLogical reasoning is a fundamental component of human intelligence and continues to be a significant challenge in the field of artificial intelligence. The growing interest in the reasoning capabilities of large language models (LLMs) highlights the pressing need for robust benchmarks and evaluation methods (Luo et al., 2023) .  \nWhile many datasets have been proposed to evaluate logical reasoning capabilities of LLMs, earlier datasets do not exclusively evaluate logical reasoning in isolation, e.g., LogiQA (Liu et al., 2020), and  \n*Equal contribution.  \n†Correspondence to: [anjiang@cs.stanford.edu](anjiang@cs.stanford.edu)  \nReClor (Yu et al., 2020), which combine logical reasoning with commonsense reasoning.  \nRecently, new datasets have been introduced to assess logical reasoning in isolation, such as FOLIO (Han et al., 2024a) and P-FOLIO (Han et al., 2024b) . These datasets are manually curated by researchers and focus on logical problems based on inference rules, which involve deriving conclusions from a set of premises.  \nIn this work, we introduce SATBench, a benchmark designed to create logical puzzles from Boolean satisfiability (SAT) problems (Cook, 1971 ; Pan et al., 2024) with LLMs. Unlike benchmarks based on inference rules, SAT problems are characterized as search-based logical reasoning tasks, where the objective is to determine a truth assignment that fulfills a specified set of logical constraints (Madusanka et al., 2024) . This approach to logical reasoning emphasizes a search process akinto backtracking used in SAT solvers. Unlike other search-based benchmarks such as ZebraLogic (Linet al., 2025), which presuppose the existence of a valid solution, SAT problems can result in either a satisfiable solution (SAT) or no solution (UNSAT) . As shown in Figure 1, starting from a SAT formula in Conjunctive Normal Form (CNF), such as (A ∨ ¬B) ∧ (¬C ∨ ¬D), our framework uses LLMs to generate a story context and define a mapping between formula variables and entities in the story. Each clause is then translated into a natural language condition based on this mapping. By sampling CNF formulas with varying numbers of clauses, we can control puzzle difficulty. To ensure the quality of resulting logical puzzles, we reverse the generation process: LLMs translate the natural language conditions back into logical formulas, whic","cbCaigWDzaGvW4Sn","https://ap.wps.com/l/cbCaigWDzaGvW4Sn","pdf",463551,18,"English","# Abstract\n# Introduction\n## Related Work\n## SATBench Methodology\n### Generation Pipeline\n### Consistency Validation\n### Evaluation Pipeline\n# Results","[{\"question\":\"What is SATBench and what does it benchmark?\",\"answer\":\"SATBench is a benchmark that evaluates the logical reasoning capabilities of large language models using puzzles generated from Boolean satisfiability (SAT) problems.\"},{\"question\":\"How are SATBench puzzles generated and how is difficulty controlled?\",\"answer\":\"Each puzzle is generated from a sampled SAT formula and translated into a story context with natural-language conditions using LLMs. Difficulty is adjusted by varying the number of clauses in the sampled CNF formulas.\"},{\"question\":\"How are puzzles validated and how is model reasoning evaluated?\",\"answer\":\"Puzzle quality is verified with both LLM-assisted and solver-based consistency checks, and reasoning traces are assessed using an LLM-as-a-judge strategy in the evaluation pipeline.\"}]","SATBench - Benchmarking LLMs’ Logical Reasoning via Automated Puzzle Generation from SAT Formulas | PDF",45]