[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-178840-105":59,"doc-detail-178840-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","2022-findings-acl-311","2022 findings - ACL 311","","Evaluation results for an ACL 311 task focused on building candidate-explanation data and comparing methods across multiple datasets and languages. The study reports dataset splits and term counts for SAT, Google, BATS, and E-KAR (Zh/En), then benchmarks embedding-based baselines, pretrained language models, and fine-tuned variants. Generation quality is assessed using metrics such as ROUGE, BERTScore, BLEURT, MoverScore, and accuracy, including human performance and gold upper bounds. It also includes example lexical and explanation pairs used for contrastive evaluation.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/2022-findings-acl-311/178840/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/2022-findings-acl-311/178840.png","ImageObject",300,407,{"name":92,"@type":93},"Aditya","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-10-08","2026-09-02",true,{"@type":102,"interactionType":103,"userInteractionCount":39},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What datasets and language variants are evaluated?","Question",{"text":112,"@type":113},"The evaluation covers SAT, Google, BATS, and E-KAR. E-KAR is reported in both Chinese (Zh) and English (En) settings with separate splits and language-specific candidate data.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"Which model families are compared in the results?",{"text":117,"@type":113},"Baselines include pre-trained word embeddings such as Word2Vec, GloVe, and FastText, followed by pretrained language models like BERT and RoBERTa variants, and then fine-tuned versions of these models. Generation-based EG methods such as BART and T5 are also evaluated for explanation outputs.",{"name":119,"@type":110,"acceptedAnswer":120},"How are explanation outputs assessed?",{"text":121,"@type":113},"Quality is measured with multiple metrics including ROUGE, BERTScore, BLEURT, and MoverScore, alongside accuracy values. Results are shown for different configurations and compared against human and gold references.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},178840,1788333925,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":39,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},962085564549,"https://ap-avatar.wpscdn.com/davatar_085a072bc5b1113ac321206ff7593b45","| Dataset Lang. |  | Data Size \\# of Terms Has\u003Cbr>(train / val / test) in Cand. Expl. |  |  |\n| --- | --- | --- | --- | --- |\n| SAT | En | 0 / 37 / 337 | 2 | 7 |\n| Google | En | 0 / 50 / 500 | 2 | 7 |\n| BATS    | En | 0 / 199 / 1,799 | 2 | 7  |\n| E-KAR\u003Cbr>| \u003Cbr>Zh | 1,155 /165 / 335 | 2 (64:5%) , 3 (35:5%) | 3 |\n|  | En | 870 / 119 / 262 | 2 (60:5%) ,\u003Cbr>3 (39:5%) | 3 |\n\n\n| Method | SAT | Google | BATS | E-KAR (H/E) |\n| --- | --- | --- | --- | --- |\n|  |  |  |  | Zh En |\n| \u003Cbr>Pre-trained Word Embeddings |  |  |  |  |\n| Word2Vecy | 41.5 | 93.2 | 63.9 | 28.2/ - 25.6/ - |\n| GloVey | 47.7 | 96.0 | 67.6 | 30.9/ - 27. 8/ - |\n| FastTexty | 47.1 | 96.6 | 72.0 | 31.4/ - 28.2/ - |\n| Pre-trained Language Models |  |  |  |  |\n| BERTyb | 32.9 | 80.8 | 61.5 | 34.5/ - 30.4/ - |\n| RoBERTayb | 42.4 | 90.8 | 69.7 | 41.7/ - 37.4/ - |\n| RoBERTayl | 45.4 | 93.4 | 72.2 | 44.6/ - 39.0/ - |\n| Fine-tuned Language Models |  |  |  |  |\n| BERTb | 38.9 | 86.6 | 68.0 | 41.8/46.7 37.9/42.2 |\n| RoBERTab | 47.7 | 93.8 | 75.2 | 46.9/51.1 42.2/48.1 |\n| RoBERTal | 51.6 | 96.9 | 78.2 | 50.1/54.8 46.7/50.5 |\n| Human | - | - | - | 77.8/83.3 |\n\n\n| EG Method |  | E-KAR (Zh) |  |  |  |  |  | E-KAR (En) |  |  |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  |  | ROUGE | BERT. | BLRT. | Mover. | Acc \" (􀀁 \\#) |  | ROUGE | BERT. | BLRT. | Mover. | Acc \" (􀀁 \\#) |\n| None (7) |  | N/A | N/A | N/A | N/A | 29.1 (68.6) |  | N/A | N/A | N/A | N/A | 25.6 (72.1) |\n| BARTb (7) |  | 39.85 | 72.68 | 63.43 | 64.72 | 33.0 (64.7) |  | 17.71 | 91.27 | 54.40 | 59.91 | 29.0 (68.7) |\n| BARTl (7) |  | 40.39 | 72.67 | 63.60 | 64.57 | 38.8 (58.9) |  | 18.34 | 91.54 | 55.48 | 60.13 | 34.1 (67.6) |\n| T5b (7) |  | 43.37 | 83.17 | 66.34 | 75.92 | 30.7 (67.0) |  | 17.44 | 91.17 | 53.71 | 60.40 | 25.6 (72.1) |\n| T5l (7) |  | - | - | - | - | - |  | 19.77 | 91.44 | 55.00 | 60.78 | 29.4 (68.3) |\n| None (3) |  | N/A | N/A | N/A | N/A | 30.5 (67.2) |  | N/A | N/A | N/A | N/A | 26.7 (71.0) |\n| BARTb (3) |  | 39.08 | 72.84 | 62.10 | 65.07 | 33.4 (64.3) |  | 25.14 | 91.85 | 56.16 | 62.16 | 29.8 (67.9) |\n| BARTl (3) |  | 39.18 | 72.93 | 62.45 | 65.13 | 36.1 (61.6) |  | 25.31 | 91.92 | 56.14 | 62.26 | 32.4 (65.3) |\n| T5b (3) |  | 40.04 | 82.52 | 63.54 | 74.99 | 34.0 (63.7) |  | 26.59 | 92.12 | 57.39 | 63.01 | 30.2 (67.5) |\n| T5l (3) |  | - | - | - | - | - |  | 28.10 | 92.38 | 58.76 | 63.64 | 31.3 (66.4) |\n| Gold |  | N/A | N/A | N/A | N/A | 97.7 (0.0) |  | N/A | N/A | N/A | N/A | 97.7 (0.0) |\n\n| 3\u003Cbr>33 | 51\u003Cbr>50\u003Cbr>46.\u003Cbr>46.\u003Cbr>9.3\u003Cbr>.3 | 72\u003Cbr>63.3\u003Cbr>.7\u003Cbr>.0\u003Cbr>7 7 | .0 |\n| --- | --- | --- | --- |\n\n| Q) | 氧气(oxygen):臭氧(ozone) |\n| --- | --- |\n| A) | 盐(salt):氯化钠(sodium chloride) |\n| B) | 硫酸(sulfuric acid):硫(sulfur) |\n| C) | 石墨(graphite):金刚石(diamond) |\n| D) | 石灰水(lime water):氢氧化钙(calcium hydroxide) |\n\n\n| EQ | 氧气和臭氧都只由氧元素组成。Both oxygen and ozone are made of only the oxygen element. |\n| --- | --- |\n| EyQ | 臭氧是氧气的一种。Ozone is a kind of oxygen. |\n| EA | 氯化钠是盐的主要成分，盐和氯化钠不是只由一种元素组成。Sodium chloride is the main component of  salt. Neither salt nor sodium chloride is made of only one element. |\n| EyA | 氯化钠是盐的一种。Sodium chloride is a kind of salt. |","cbCaip75zm8dfFI7","https://ap.wps.com/l/cbCaip75zm8dfFI7","pdf",1219513,15,"English","# Dataset overview\n## Candidate-explanation splits\n# Model and embedding baselines\n## Word embeddings\n## Pretrained language models\n## Fine-tuned language models\n# Evaluation for E-KAR (Zh) and E-KAR (En)\n## Metrics and accuracy\n# Example explanation pairs","[{\"question\":\"What datasets and language variants are evaluated?\",\"answer\":\"The evaluation covers SAT, Google, BATS, and E-KAR. E-KAR is reported in both Chinese (Zh) and English (En) settings with separate splits and language-specific candidate data.\"},{\"question\":\"Which model families are compared in the results?\",\"answer\":\"Baselines include pre-trained word embeddings such as Word2Vec, GloVe, and FastText, followed by pretrained language models like BERT and RoBERTa variants, and then fine-tuned versions of these models. Generation-based EG methods such as BART and T5 are also evaluated for explanation outputs.\"},{\"question\":\"How are explanation outputs assessed?\",\"answer\":\"Quality is measured with multiple metrics including ROUGE, BERTScore, BLEURT, and MoverScore, alongside accuracy values. Results are shown for different configurations and compared against human and gold references.\"}]","2022 findings - ACL 311 | PDF",38]