[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-191387-105":53,"doc-detail-191387-en":126},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":119,"head_meta":121,"extra_data":123,"updated_unix":125},105,"en","instructds","InstructDS","","InstructDS is evaluated on dialogue summarization and related generation settings using multiple datasets (SAMSum, DialogSum, TODSum, and DREAM) and several quality metrics. The document reports train/validation/test sizes, QDS triples, and performance comparisons across baseline language models (e.g., Alpaca, Flan variants, ChatGPT) and summarization-focused architectures (e.g., BART-based approaches). Results include ROUGE-1/2/L, BS scores, multi-choice accuracy, and human-annotator ratings for faithfulness, fluency, informativeness, and conciseness, including a setting with reference summary length.",{"@graph":63,"@context":118},[64,80,101],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/instructds/191387/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/instructds/191387.png","ImageObject",442,249,{"name":88,"@type":89},"Theodora","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-27","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":76},"InteractionCounter",{"@type":100},"ViewAction",{"@type":102,"mainEntity":103},"FAQPage",[104,110,114],{"name":105,"@type":106,"acceptedAnswer":107},"Which datasets and splits are used to evaluate InstructDS?","Question",{"text":108,"@type":109},"The evaluation includes SAMSum, DialogSum, TODSum, and DREAM, with reported train, validation, and test sizes plus QDS triples for relevant datasets.","Answer",{"name":111,"@type":106,"acceptedAnswer":112},"What automatic metrics are reported in the document?",{"text":113,"@type":109},"The document reports ROUGE-1, ROUGE-2, ROUGE-L, and BS scores, along with a multi-choice accuracy metric.",{"name":115,"@type":106,"acceptedAnswer":116},"How is InstructDS evaluated by humans?",{"text":117,"@type":109},"Human evaluation compares systems on faithfulness, fluency, informativeness, and conciseness, with scores reported for both human-written and model-generated outputs.","https://schema.org",{"og:url":78,"og:type":120,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":122,"canonical":78},"index,follow",{"doc_id":124,"site_id":56},191387,1788408324,{"code":4,"msg":5,"data":127},{"doc_id":124,"user_id":128,"nickname":88,"user_avatar":129,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":130,"file_id":131,"file_url":132,"file_type":133,"file_size":134,"view_count":76,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":135,"language":136,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":137,"faqs":138,"seo_title":139,"seo_description":61,"update_tm":125,"read_time":140},687197207919,"https://ap-avatar.wpscdn.com/avatar/a000253d6f5f7c60be?x-image-process=image/resize,m_fixed,w_180,h_180&k=1779446848396160552","| Dataset | \\# Train | \\# Validation | \\# Test | \\# QDS Triples | Direct Exposure Alpaca Flan-Series InstructDS |\n| --- | --- | --- | --- | --- | --- |\n| SAMSum (Gliwa et al., 2019) | 14,732 | 818 | 819 | 18,245 | ✗ ✓ ✓ |\n| DialogSum (Chen et al., 2021) | 12,460 | 500 | 1,500 | 18,600 | ✗ ✗ ✓ |\n| TODSum (Zhao et al., 2021) | 7,892 | 999 | 999 | 8,705 | ✗ ✗ ✓ |\n\n\n| DREAM (Sun et al., 2019) | 6,116 | 2,040 | 2,041 | - | ✗ ✓ ✗ |\n| --- | --- | --- | --- | --- | --- |\n\n| Models Params |  | ROUGE-1 |  |  | ROUGE-2 |  |  | ROUGE-L |  |  | BS |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  |  | F1 Pre Rec |  |  | F1 Pre Rec |  |  | F1 Pre Rec |  |  |  |\n| Pointer-Generator BART\u003Cbr>MV-BART Coref-BART\u003Cbr>ConDigSum GPT-3-finetune | -\u003Cbr>400M\u003Cbr>400M\u003Cbr>400M\u003Cbr>400M\u003Cbr>175B∗ | 40.1\u003Cbr>53.0\u003Cbr>53.9\u003Cbr>53.7\u003Cbr>54.3 | -\u003Cbr>59.0\u003Cbr>55.7\u003Cbr>56.9\u003Cbr>56.0\u003Cbr>- | -\u003Cbr>52.8\u003Cbr>57.4\u003Cbr>56.4\u003Cbr>57.6\u003Cbr>- | 15.3\u003Cbr>28.4\u003Cbr>28.4\u003Cbr>28.5\u003Cbr>29.3\u003Cbr>29.8 | -\u003Cbr>32.1\u003Cbr>29.3\u003Cbr>30.5\u003Cbr>30.4\u003Cbr>- | -\u003Cbr>28.2\u003Cbr>30.6\u003Cbr>29.7\u003Cbr>31.2\u003Cbr>- | 36.6\u003Cbr>44.2\u003Cbr>44.4\u003Cbr>44.3\u003Cbr>45.2\u003Cbr>45.9 | -\u003Cbr>49.3\u003Cbr>45.7\u003Cbr>46.9\u003Cbr>46.6\u003Cbr>- | -\u003Cbr>44.0\u003Cbr>47.5\u003Cbr>46.5\u003Cbr>48.0\u003Cbr>- | -\u003Cbr>53.3\u003Cbr>53.6\u003Cbr>53.5\u003Cbr>54.0\u003Cbr>- |\n|  |  | 53.4 |  |  |  |  |  |  |  |  |  |\n| Alpaca | 7B | 28.2 | 26.0 | 39.8 | 5.7 | 5.1 | 8.3 | 20.5 | 19.2 | 29.0 | 19.4 |\n| Flan-T5-XXL | 11B | 52.6 | 62.6 | 50.0 | 28.5 | 34.1 | 27.1 | 44.1 | 52.5 | 41.9 | 53.2 |\n| Flan-UL2 | 20B | 53.3 | 60.3 | 52.5 | 28.0 | 32.0 | 27.7 | 44.1 | 50.0 | 43.3 | 53.5 |\n| ChatGPT | 175B | 32.7 | 22.4 | 70.2 | 12.3 | 8.4 | 27.1 | 24.7 | 16.9 | 53.6 | 32.5 |\n| InstructDS | 3B∗ | 55.3 | 58.8 | 57.5 | 31.3 | 33.5 | 32.6 | 46.7 | 49.7 | 48.6 | 55.5 |\n| w/ reference summary length |  |  |  |  |  |  |  |  |  |  |  |\n| ChatGPT | 175B | 40.8 | 39.3 | 43.4 | 13.7 | 13.2 | 14.6 | 31.5 | 30.5 | 33.4 | 40.0 |\n| InstructDS | 3B∗ | 58.4 | 58.5 | 58.8 | 32.8 | 32.9 | 33.0 | 48.9 | 49.0 | 49.2 | 58.5 |\n\n\n| Models | DialogSum |  |  |  | TODSum |  |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | R-1 | R-2 | R-L | BS | R-1 | R-2 | R-L | BS |\n| Alpaca | 25.5 | 4.9 | 18.8 | 18.0 | 33.6 | 6.9 | 21.8 | 14.6 |\n| Flan-T5-Large | 38.8 | 14.4 | 30.9 | 38.7 | 37.3 | 13.4 | 25.3 | 23.6 |\n| Flan-T5-XXL | 39.3 | 15.8 | 32.4 | 39.5 | 39.3 | 14.2 | 27.2 | 23.5 |\n| Flan-UL2 | 40.8 | 16.5 | 33.3 | 40.9 | 41.6 | 14.6 | 27.9 | 24.3 |\n| ChatGPT | 38.4 | 12.9 | 29.8 | 38.8 | 39.8 | 11.8 | 24.5 | 24.9 |\n| BART | 47.3 | 21.3 | 38.6 | 45.8 | 73.1 | 56.8 | 64.0 | 64.3 |\n| InstructDS | 47.8 | 22.2 | 39.4 | 47.0 | 89.3 | 78.9 | 85.4 | 85.5 |\n\n\n| Models | Multi-Choice Acc. |\n| --- | --- |\n| Random | 33.3% |\n| Alpaca\u003Cbr>Flan-T5-Large\u003Cbr>Flan-T5-XXL\u003Cbr>Flan-UL2 ChatGPT | 51.3%\u003Cbr>53.1%\u003Cbr>58.5%\u003Cbr>56.8%\u003Cbr>60.8% |\n| InstructDS\u003Cbr>+ In-domain | 57.8%\u003Cbr>65.9% |\n\n| Models | Human Annotator |  |  |  | ChatGPT |  |  |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | Faithfulness | Fluency | Informativeness | Conciseness | Faithfulness | Fluency | Informativeness | Conciseness |\n| BART | 3.85 (1.3) | 4.36 (0.8) | 3.22 (1.0) | 4.30 (0.9) | 4.22 (1.1) | 4.80 (0.5) | 3.37 (1.0) | 4.93 (0.3) |\n| Alpaca | 3.24 (1.3) | 3.77 (1.3) | 3.45 (1.1) | 3.11 (1.4) | 3.59 (1.3) | 4.07 (1.0) | 3.19 (1.2) | 4.29 (1.0) |\n| Flan-UL2 | 4.00 (1.3) | 4.38 (0.9) | 3.03 (1.2) | 4.29 (1.0) | 4.45 (0.9) | 4.78 (0.5) | 3.52 (1.0) | 4.91 (0.3) |\n| ChatGPT | 4.52 (0.9) | 4.38 (0.9) | 4.62 (0.6) | 2.77 (1.4) | 4.94 (0.3) | 4.94 (0.2) | 4.78 (0.4) | 4.89 (0.3) |\n| Human-written | 4.34 (1.0) | 4.54 (0.7) | 3.58 (1.1) | 4.36 (0.9) | 4.49 (0.8) | 4.81 (0.4) | 3.74 (1.0) | 4.95 (0.3) |\n| InstructDS 4.13 (1 . 1) |  | 4.35 (0.8) | 3.54 (1.0) | 4.23 (1 .0) 4.60 (0 . 8) |  | 4.82 (0.4) | 3.78 (0.9) | 4.92 (0.3) |","cbCaivUxw8eJkDUf","https://ap.wps.com/l/cbCaivUxw8eJkDUf","pdf",889827,24,"English","# Dataset and Split Statistics\n# ROUGE and BS Results\n# Alpaca/Flan/ChatGPT vs InstructDS Comparison\n# Multi-Choice Accuracy\n# Human Annotator Evaluation","[{\"question\":\"Which datasets and splits are used to evaluate InstructDS?\",\"answer\":\"The evaluation includes SAMSum, DialogSum, TODSum, and DREAM, with reported train, validation, and test sizes plus QDS triples for relevant datasets.\"},{\"question\":\"What automatic metrics are reported in the document?\",\"answer\":\"The document reports ROUGE-1, ROUGE-2, ROUGE-L, and BS scores, along with a multi-choice accuracy metric.\"},{\"question\":\"How is InstructDS evaluated by humans?\",\"answer\":\"Human evaluation compares systems on faithfulness, fluency, informativeness, and conciseness, with scores reported for both human-written and model-generated outputs.\"}]","InstructDS | PDF",8]