[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-179003-105":59,"doc-detail-179003-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","2024nlp4science-120-syllogism-reasoning-datasets","2024.nlp4science-1.20 - syllogism reasoning datasets","","Syllogism reasoning and conclusion validity evaluation are summarized through multiple datasets, including SylloFigure, Avicenna, Reasoning, and FOLIO-style first-order logic settings. The material compares premise-conclusion configurations and quantifier types (universal affirmative/negative, particular affirmative/negative) while reporting validity identification or selection accuracy using models such as GPT-2, RoBERTa, PaLM 2, GPT-3.5, and GPT-4/Lite variants. Tables present distributional breakdowns, dataset sizes, and performance scores across figures and configurations.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/2024nlp4science-120-syllogism-reasoning-datasets/179003/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/2024nlp4science-120-syllogism-reasoning-datasets/179003.png","ImageObject",300,407,{"name":92,"@type":93},"Rowan","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-18","2026-09-02",true,{"@type":102,"interactionType":103,"userInteractionCount":14},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What logical proposition types are covered in the material?","Question",{"text":112,"@type":113},"It lists universal affirmative (All S are P), universal negative (No S is P), particular affirmative (Some S is P), and particular negative (Some S is not P).","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does the document structure premise and conclusion configurations?",{"text":117,"@type":113},"It uses a figure-based setup mapping major and minor premises to a conclusion in S-P form, with variations across figures 1–4 and corresponding M-P/S-M or M-S relations.",{"name":119,"@type":110,"acceptedAnswer":120},"Which datasets and model systems are compared for conclusion validity?",{"text":121,"@type":113},"Datasets include SylloBASE/SylloFigure, Avicenna, Reasoning, and FOLIO, with evaluation pipelines using models such as RoBERTa, PaLM 2-L, GPT-3.5, PaLM 2, and Logic-LM (GPT-4), reporting validity identification or selection performance.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},179003,1788334892,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":14,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":52,"language":139,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":67,"update_tm":129,"read_time":143},1099514067415,"https://ap-avatar.wpscdn.com/avatar/100002539d78ffe74a7?x-image-process=image/resize,m_fixed,w_180,h_180&k=1779092875211072502","| Proposition | Type | Gen. quant. |\n| --- | --- | --- |\n| All S are P. | Universal Affirmative (A) | S ⊆ P |\n| No S is P. | Universal Negative (E) | S ∩ P = ∅ |\n| Some S is P. | Particular Affirmative (I) | S ∩ P  ∅ |\n| Some S is not P. | Particular Negative (O) | S − P  ∅ |\n\n\n| Figure | 1 | 2 | 3 | 4 |\n| --- | --- | --- | --- | --- |\n| Major Premise | M-P | P-M | M-P | P-M |\n| Minor Premise | S-M | S-M | M-S | M-S |\n| Conclusion | S-P | S-P | S-P | S-P |\n\n\n| Avicenna\u003Cbr>(Aghahadi and Talebpour, 2022) | Crowdsourcing | Books, articles, etc. |  | Middle |   valid, invalid\u003Cbr>|  | Conclusion\u003Cbr>generation | GPT-2 trans. learning | 32.0% | 6,000 | Yes |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| SylloBASE\u003Cbr>(Wu et al., 2023) | Template w/\u003Cbr>GPT-3 rewrite | Wikidata ConceptNet |  | |   valid, invalid |  | Conclusion selection | RoBERTa | 72.8% | 51,000 | No |\n| Logical\u003Cbr>(Lampinen et al., 2023) | Human authored questions |  |  | |  | valid\u003Cbr>belief-consistent | Conclusion validity identification | PaLM 2-L | ∼90%\u003Cbr>(support) | 48 | No |\n| NeuBAROCO\u003Cbr>(Ando et al., 2023) | BAROCO (originally designed for human intell. test) |  |  | | | entail, contra, neu inference types | Conclusion validity identification | GPT-3.5 | 51.7%\u003Cbr>(overall) | 375 | No |\n| Reasoning\u003Cbr>(Eisape et al., 2024) | Template | Hand-crafted triples list |  | |   valid, invalid |  | Conclusion selection | PaLM 2 | ∼75% | 1,920 | Yes |\n| FOLIO\u003Cbr>(Han et al., 2022) | Template w/ crowd-sourcing rewrite |  | N/A | | First-order Logic Datasets\u003Cbr>  true, false, unknown |  | Conclusion truth identification | Logic-LM\u003Cbr>(GPT-4) | 78.1% | 1,435 | Yes |\n| ProntoQA\u003Cbr>(Saparov and He, 2023) | Template | Generated\u003Cbr>ontology |  | |  | true,\u003Cbr>false | Validity of sorites | GPT-3 | ∼90% | 400 | Yes |\n\n|  |  | SylloFigure | Avicenna | Reasoning |\n| --- | --- | --- | --- | --- |\n| Standard (%) |  | 0.9 | 0.6 | 100 |\n| Singular (%) |  | 64.7 | 27.2 | 0 |\n| Proposition | Condition (%)\u003Cbr>Exclusive (%) | 2.3 | 9.5 | 0 |\n|  |  | 0.1 | 1.0 | 0 |\n| Others (%) Total |  | 32.0 | 61.7 | 0 |\n|  |  | 2,448 | 1,864 | 2,560 |\n| Configuration | Coverage (%) Actual count\u003Cbr>Syllo assessed (%) | >4.3\u003Cbr>> 11 | >2.7\u003Cbr>>7 | 100\u003Cbr>256 |\n|  |  | 71.1 | 60.9 | 100 |\n| Total syllogisms |  | 868 | 622 | 2,560 |\n\n\n| Dataset | \\# | GPT-4 | GPT-4o |\n| --- | --- | --- | --- |\n| SylloFigure | 868 | 74.3 | 70.2 |\n| Avicenna | 622 | 72.5 | 53.4 |\n| Reasoning | 2,560 | 90.2 | 95.4 |\n\n| 0.1 |  | 0.4 |  |  |  |  |  | 0.1 |  | 0.6 |  |  |  | 0.6 |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | 0.5 |  | 0.1 |  | 0.1 |  | 0.3 |  | 0.2 |  |  |  | 0.2 |  | 0.7 |\n|  |  |  |  |  |  |  |  |  |  | 0.3 |  |  |  | 0.1 |  |\n|  |  |  |  |  | 0.1 |  | 0.4 |  |  |  |  |  |  |  |  |\n|  |  |  | 0.4 |  | 0.3 |  | 0.1 |  | 0.4 | 0.1 |  |  | 0.6 |  | 0.3 |\n|  |  |  |  |  | 0.2 |  |  |  |  |  |  |  | 0.1 |  |  |\n|  | 0.2 |  |  |  | 0.1 |  |  |  |  |  | 0.2 |  | 0.3 |  | 0.4 |\n|  |  |  | 0.3 |  | 0.2 |  | 0.2 |  |  |  | 0.1 |  |  |  |  |\n|  |  | 0.7 |  |  |  |  |  |  |  | 0.1 |  |  |  | 0.1 | 0.1 |\n|  | 0.1 |  | 0.1 |  | 0.1 |  |  |  | 0.1 |  |  |  | 0.1 |  |  |\n|  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n|  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n|  |  |  | 0.3 |  | 0.1 |  |  |  | 0.1 |  | 0.2 |  |  |  | 0.2 |\n|  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n|  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n|  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n\n| Figure | SylloFigure |  |  |  | Avicenna |  |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | Mood | \\# | GPT-4 | GPT-4o | Mood | \\# | GPT-4 | GPT-4o |\n| 1 | AAA | 47 | 0.21 | 0.28 | AAA | 310 | 0.20 | 0.42 |\n|  | AAI | 38 | 0.32 | 0.42 | AAI | 12 | 0.33 | 0.42 |\n|  | AII | 502 | 0.21 | 0.26 | AII | 25 | 0.28 | 0.68 |\n|  | N/A | 56 | 0.34 | 0.32 | EAE | 2 | 1 | 0.50 |\n| 2 | EAE | 1 | 0 | 0 | EAE | 3 | 0 | 0 |\n|  | N/A | 180 | 0.28 | 0.36 ","cbCaik35qlEPoNSL","https://ap.wps.com/l/cbCaik35qlEPoNSL","pdf",242997,"English","# Proposition and quantifier types\n## Universal and particular forms\n# Premise-to-conclusion configuration\n## Major/minor premise mapping\n# Dataset overview and model comparison\n## Avicenna, SylloBASE, Logical, NeuBAROCO, Reasoning, FOLIO, ProntoQA\n# Performance and distribution analysis\n## Validity accuracy by dataset and model","[{\"question\":\"What logical proposition types are covered in the material?\",\"answer\":\"It lists universal affirmative (All S are P), universal negative (No S is P), particular affirmative (Some S is P), and particular negative (Some S is not P).\"},{\"question\":\"How does the document structure premise and conclusion configurations?\",\"answer\":\"It uses a figure-based setup mapping major and minor premises to a conclusion in S-P form, with variations across figures 1–4 and corresponding M-P/S-M or M-S relations.\"},{\"question\":\"Which datasets and model systems are compared for conclusion validity?\",\"answer\":\"Datasets include SylloBASE/SylloFigure, Avicenna, Reasoning, and FOLIO, with evaluation pipelines using models such as RoBERTa, PaLM 2-L, GPT-3.5, PaLM 2, and Logic-LM (GPT-4), reporting validity identification or selection performance.\"}]","2024.nlp4science-1.20 - syllogism reasoning datasets | PDF",25]