[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-178957-105":3,"detail-sidebar-cat-0-en-105":81,"doc-detail-178957-en":130},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":74,"head_meta":76,"extra_data":78,"updated_unix":80},105,"en","language-to-logic-multi-choice-inference-evaluation","Language to Logic  - Multi-choice inference evaluation","","This document presents an evaluation of language-to-reasoning inference methods across multiple benchmark datasets and task settings. It compares model variants such as LLaMA-based base and logicot-tuned systems on metrics like accuracy across domains and language variants (including bilingual LQ vs LQ zh). Results are reported for one-step inference, inference chains, and multi-choice tasks, with additional removal experiments quantifying contributions of components by tracking performance shifts on LogiEval and MMLU.",{"@graph":14,"@context":73},[15,34,56],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/research-report/","Research & Report",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/language-to-logic-multi-choice-inference-evaluation/178957/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/language-to-logic-multi-choice-inference-evaluation/178957.png","ImageObject",300,407,{"name":42,"@type":43},"Rizky","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-10-03","2026-09-02",true,{"@type":52,"interactionType":53,"userInteractionCount":55},"InteractionCounter",{"@type":54},"ViewAction",5,{"@type":57,"mainEntity":58},"FAQPage",[59,65,69],{"name":60,"@type":61,"acceptedAnswer":62},"What inference tasks are evaluated in this document?","Question",{"text":63,"@type":64},"The evaluation covers language-to-logic tasks including Language to Logic, One-Step Inference, Inference Chain, and Multi-choice.","Answer",{"name":66,"@type":61,"acceptedAnswer":67},"Which models are compared on the benchmarks?",{"text":68,"@type":64},"The document compares LLaMA-7b-base, LLaMA-30b-supercot, and a logic-tuned variant (LLaMA-7b-logicot, with additional rows referencing ChatGPT/GPT-4 in some tables).",{"name":70,"@type":61,"acceptedAnswer":71},"How are results analyzed beyond overall scores?",{"text":72,"@type":64},"Performance is broken down by datasets (including LQ zh and OOD settings), and by subject domains, then summarized with ablation-style removals and overall checks using LogiEval and MMLU.","https://schema.org",{"og:url":32,"og:type":75,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":77,"canonical":32},"index,follow",{"doc_id":79,"site_id":7},178957,1788334632,{"code":4,"msg":82,"data":83},"success",[84,88,92,96,100,105,110,114,119,122,126],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":85,"show_sort_weight":86,"slug":87},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":89,"show_sort_weight":90,"slug":91},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":93,"show_sort_weight":94,"slug":95},"Exam",70,"exam",{"id":55,"doc_module":4,"doc_module_name":25,"category_name":97,"show_sort_weight":98,"slug":99},"Comic",60,"comic",{"id":101,"doc_module":4,"doc_module_name":25,"category_name":102,"show_sort_weight":103,"slug":104},6,"Technology",50,"technology",{"id":106,"doc_module":4,"doc_module_name":25,"category_name":107,"show_sort_weight":108,"slug":109},7,"Healthcare",40,"healthcare",{"id":111,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":112,"slug":113},8,30,"research-report",{"id":115,"doc_module":4,"doc_module_name":25,"category_name":116,"show_sort_weight":117,"slug":118},9,"Religion & Spirituality",20,"religion-spirituality",{"id":117,"doc_module":4,"doc_module_name":25,"category_name":120,"show_sort_weight":117,"slug":121},"World Cup","world-cup",{"id":123,"doc_module":4,"doc_module_name":25,"category_name":124,"show_sort_weight":123,"slug":125},10,"Lifestyle","lifestyle",{"id":127,"doc_module":4,"doc_module_name":25,"category_name":128,"show_sort_weight":55,"slug":129},19,"General","general",{"code":4,"msg":82,"data":131},{"doc_id":79,"user_id":132,"nickname":42,"user_avatar":133,"doc_module":4,"category_id":111,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":55,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":139,"language":140,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":12,"update_tm":80,"read_time":144},962085564807,"https://ap-avatar.wpscdn.com/davatar_6f874abed73319feea01a86fa6f0fab8","| Task | Origin | Size |\n| --- | --- | --- |\n| Language to Logic | LOGICINFERENCE & FOLIO | 13,206 |\n| One-Step Inference | LOGICINFERENCE & FOLIO | 23,943 |\n| Inference Chain | LOGICINFERENCE & EntailmentBank | 26,228 |\n| Multi-choice | LogiQA & ReClor | 5,606 |\n\n\n| Dataset | LQ | LQ zh | RC | AL | LQ ood | CT | HL | TN | Overall |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| Size | 1572 | 1594 | 500 | 230 | 1354 | 805 | 35891 | 10071 | 52017 |\n| LLaMA-7b-base | 18.04 | 19.06 | 15.83 | 13.91 | 20.25 | 32.40 | 25.20 | 37.35 | 20.22 |\n| LLaMA-30b-supercot | 19.31 | 26.35 | 17.81 | 17.98 | 18.41 | 24.10 | 32.26 | 41.91 | 24.78 |\n| Falcon-40b-instruct LLaMA-7b-logicot | 23.21\u003Cbr>50.25 | 19.77\u003Cbr>32.77 | 26.77\u003Cbr>57.60 | 12.70\u003Cbr>16.96 | 17.33\u003Cbr>38.79 | 16.13\u003Cbr>35.68 | 28.49\u003Cbr>35.44 | 44.66\u003Cbr>58.05 | 23.63\u003Cbr>40.69 |\n\n| Dataset | Target |\n| --- | --- |\n| LogiQA 2.0 test | 4-way multi-choice |\n| LogiQA 2.0 zh test | 4-way multi-choice |\n| ReClor dev | 4-way multi-choice |\n| AR-LSAT test | 5-way multi-choice |\n| LogiQA 2.0 OOD | 4-way multi-choice |\n| ConTRoL test HELP test TaxiNLI test | E, C, N\u003Cbr>E, C, N\u003Cbr>E, C, N |\n\n\n| Task | LLaMA-7b-base | LLaMA-7b-logicot |\n| --- | --- | --- |\n| Math | 25.1 | 29.0 |\n| Health | 34.0 | 42.9 |\n| Physics | 29.4 | 34.2 |\n| Business | 34.8 | 57.2 |\n| Biology | 32.4 | 46.5 |\n| Chemistry | 25.4 | 33.0 |\n| Computer science | 28.2 | 40.0 |\n| Economics | 26.5 | 38.5 |\n| Engineering | 25.5 | 32.4 |\n| Philosophy | 30.8 | 37.6 |\n| Other | 40.4 | 50.8 |\n| History | 38.2 | 55.2 |\n| Geography | 28.8 | 52.5 |\n| Politics | 32.1 | 53.4 |\n| Psychology | 33.2 | 50.9 |\n| Culture | 37.3 | 56.9 |\n| Law | 29.3 | 39.6 |\n| STEM | 27.6 | 34.8 |\n| Humanities | 31.6 | 41.8 |\n| Social Sciences | 31.5 | 49.2 |\n| Other (misc.) | 36.4 | 47.7 |\n| Average | 31.8 | 43.3 |\n\n\n| Dataset | LQ | LQ zh | RC | AL | LQ ood | CT | HL | TN | Overall |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n| LLaMA-7b-logicot ChatGPT\u003Cbr>GPT-4 | 50.25\u003Cbr>52.37\u003Cbr>72.25 | 32.77\u003Cbr>53.18\u003Cbr>70.56 | 57.60 | 16.96\u003Cbr>20.42\u003Cbr>33.48 | 38.79\u003Cbr>38.44\u003Cbr>58.49 | 35.68\u003Cbr>58.45\u003Cbr>56.40 | 35.44\u003Cbr>42.13 | 58.05\u003Cbr>57.30\u003Cbr>60.08 | 40.69\u003Cbr>47.46\u003Cbr>60.56 |\n|  |  |  | 57.38\u003Cbr>87.20 |  |  |  |  |  |  |\n|  |  |  |  |  |  |  | 46.01 |  |  |\n\n\n| Removed | LogiEval | MMLU |\n| --- | --- | --- |\n| None (Full data) | 40.7 | 43.3 |\n| Language to Logic | 32.4 | 38.5 |\n| One-step Inference | 38.1 | 37.7 |\n| Inference Chain | 30.8 | 35.0 |\n| Multi-choice | 35.6 | 30.9 |","cbCaihAWABXbWiqm","https://ap.wps.com/l/cbCaihAWABXbWiqm","pdf",518859,14,"English","# Task Overview\n## One-step and Chain Inference\n## Multi-choice Settings\n# Datasets and Metrics\n## Cross-language and OOD Tests\n## Domain-wise Performance\n# Ablation: Removed Components\n## Impact on LogiEval and MMLU","[{\"question\":\"What inference tasks are evaluated in this document?\",\"answer\":\"The evaluation covers language-to-logic tasks including Language to Logic, One-Step Inference, Inference Chain, and Multi-choice.\"},{\"question\":\"Which models are compared on the benchmarks?\",\"answer\":\"The document compares LLaMA-7b-base, LLaMA-30b-supercot, and a logic-tuned variant (LLaMA-7b-logicot, with additional rows referencing ChatGPT/GPT-4 in some tables).\"},{\"question\":\"How are results analyzed beyond overall scores?\",\"answer\":\"Performance is broken down by datasets (including LQ zh and OOD settings), and by subject domains, then summarized with ablation-style removals and overall checks using LogiEval and MMLU.\"}]","Language to Logic  - Multi-choice inference evaluation | PDF",35]