[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-1-en-105":3,"doc-seo-189729-105":53,"doc-detail-189729-en":127},{"code":4,"msg":5,"data":6},0,"success",[7,14,19,24,29,34,39,44,49],{"id":8,"doc_module":9,"doc_module_name":10,"category_name":11,"show_sort_weight":12,"slug":13},11,1,"Template","Presentations",90,"presentations",{"id":15,"doc_module":9,"doc_module_name":10,"category_name":16,"show_sort_weight":17,"slug":18},12,"Resumes",80,"resumes",{"id":20,"doc_module":9,"doc_module_name":10,"category_name":21,"show_sort_weight":22,"slug":23},14,"Invoices",70,"invoices",{"id":25,"doc_module":9,"doc_module_name":10,"category_name":26,"show_sort_weight":27,"slug":28},15,"Posters",60,"posters",{"id":30,"doc_module":9,"doc_module_name":10,"category_name":31,"show_sort_weight":32,"slug":33},16,"Social Media",50,"social-media",{"id":35,"doc_module":9,"doc_module_name":10,"category_name":36,"show_sort_weight":37,"slug":38},17,"Forms",40,"forms",{"id":40,"doc_module":9,"doc_module_name":10,"category_name":41,"show_sort_weight":42,"slug":43},18,"Letters",30,"letters",{"id":45,"doc_module":9,"doc_module_name":10,"category_name":46,"show_sort_weight":47,"slug":48},21,"Paper Templates",5,"papers-templates",{"id":50,"doc_module":9,"doc_module_name":10,"category_name":51,"show_sort_weight":4,"slug":52},158,"General","general-158",{"code":4,"msg":54,"data":55},"ok",{"site_id":56,"language":57,"slug":58,"title":59,"keywords":60,"description":61,"schema_data":62,"social_meta":120,"head_meta":122,"extra_data":124,"updated_unix":126},105,"en","acl-long-2023-model-fine-tuning-evaluation","ACL Long 2023 - Model Fine-tuning Evaluation","","Model fine-tuning results are presented with comparisons across architectures and prompting regimes, including fine-tuned T5 variants and ByT5 with both frozen and few-shot settings. Performance is reported using bucketed frequency ranges and Top-1% style thresholds, alongside language-specific outcomes for Arabic, Chinese, English, Finnish, Korean, Russian, and Thai. Additional analysis categorizes error types such as semantic, homophone, glyph additions/drops, and text corruption behaviors, enabling qualitative and quantitative evaluation of generation quality across languages.",{"@graph":63,"@context":119},[64,80,102],{"@type":65,"itemListElement":66},"BreadcrumbList",[67,71,74,77],{"item":68,"name":69,"@type":70,"position":9},"https://docshare.wps.com","Home","ListItem",{"item":72,"name":10,"@type":70,"position":73},"https://docshare.wps.com/template/",2,{"item":75,"name":51,"@type":70,"position":76},"https://docshare.wps.com/template/general/",3,{"item":78,"name":59,"@type":70,"position":79},"https://docshare.wps.com/template/acl-long-2023-model-fine-tuning-evaluation/189729/",4,{"url":78,"name":59,"@type":81,"image":82,"author":87,"headline":59,"publisher":90,"fileFormat":93,"inLanguage":57,"description":61,"dateModified":94,"datePublished":95,"encodingFormat":93,"isAccessibleForFree":96,"interactionStatistic":97},"DigitalDocument",{"url":83,"@type":84,"width":85,"height":86},"https://docshare.wps.com/thumbnails/acl-long-2023-model-fine-tuning-evaluation/189729.png","ImageObject",442,249,{"name":88,"@type":89},"นรินทร์","Person",{"url":68,"name":91,"@type":92},"DocShare","Organization","application/pdf","2026-09-22","2026-09-03",true,{"@type":98,"interactionType":99,"userInteractionCount":101},"InteractionCounter",{"@type":100},"ViewAction",6,{"@type":103,"mainEntity":104},"FAQPage",[105,111,115],{"name":106,"@type":107,"acceptedAnswer":108},"Which models and training settings are compared in the document?","Question",{"text":109,"@type":110},"The document compares fine-tuned T5 variants with ByT5, including settings with a frozen encoder and few-shot prompting.","Answer",{"name":112,"@type":107,"acceptedAnswer":113},"How is performance reported across different groups of data?",{"text":114,"@type":110},"Performance is shown using bucketed frequency ranges (e.g., Top 1%, 1–10%, 10–20%, 20–30%, Bottom 50%) with corresponding scores for different model variants.",{"name":116,"@type":107,"acceptedAnswer":117},"What kinds of generation errors are analyzed?",{"text":118,"@type":110},"The document categorizes errors into types including semantic errors, homophone substitutions, added glyphs, dropped glyphs, repeat glyphs, and other text-shape or missing-text issues.","https://schema.org",{"og:url":78,"og:type":121,"og:title":59,"og:site_name":91,"og:description":61},"article",{"robots":123,"canonical":78},"index,follow",{"doc_id":125,"site_id":56},189729,1788398820,{"code":4,"msg":5,"data":128},{"doc_id":125,"user_id":129,"nickname":88,"user_avatar":130,"doc_module":9,"category_id":50,"category_name":51,"doc_title":59,"doc_description":61,"doc_content":131,"file_id":132,"file_url":133,"file_type":134,"file_size":135,"view_count":101,"is_deleted":4,"is_public":9,"is_downloadable":9,"audit_status":9,"page_count":136,"language":137,"language_code":57,"site_id":56,"html_lang":57,"table_of_contents":138,"faqs":139,"seo_title":140,"seo_description":61,"update_tm":126,"read_time":141},2336475104957,"https://ap-avatar.wpscdn.com/avatar/22000c4c6bd8a5076e1?x-image-process=image/resize,m_fixed,w_180,h_180&k=1787554080175789136","| Frequency |  | Fine-tuned, wi\u003Cbr>T5 |  |  |  | th frozen encoder\u003Cbr>ByT5 |  |  |  | Fine-tuned, all p\u003Cbr>T5 |  |  |  | arameters trained\u003Cbr>ByT5 |  |  |  | Few-shot\u003Cbr>PaLM |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  |  |  | L | XL | XXL | B | L | XL | XXL | B | L | XL | XXL | B | L | XL | XXL | 8B | 62B 540B |\n| Bucket |  | B |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |\n| Top 1% | 14 | | \u003Cbr>12 | \u003Cbr>50 | \u003Cbr>66 | \u003Cbr>97 | 95 | \u003Cbr>97 | \u003Cbr>98 | \u003Cbr>36 | \u003Cbr>46 | \u003Cbr>62 | \u003Cbr>68 | \u003Cbr>99 | \u003Cbr>100 | \u003Cbr>100 | \u003Cbr>100 | \u003Cbr>84 | \u003Cbr>99 100 |\n| 1–10% |  | 29 | 24 | 67 | 69 | 97 | 95 | 98 | \u003Cbr>98 | 67 | 72 | 82 | \u003Cbr>85 | 100 | 100 | 100 | \u003Cbr>100 | 62 | \u003Cbr>98 99 |\n| 10–20% |  | 35 | 27 | 73 | 73 | 96 | 94 | 98 | \u003Cbr>98 | 74 | 79 | 89 | 91 | 100 | 100 | 100 | \u003Cbr>100 | 70 | \u003Cbr>97 99 |\n| 20–30% |  | 32 | 24 | 68 | 68 | 96 | 94 | 99 | \u003Cbr>98 | 74 | 78 | 87 | 90 | 100 | 100 | 100 | \u003Cbr>100 | 71 | \u003Cbr>97 99 |\n| Bottom 50% | \u003Cbr>2 | 9 | 22 | 64 | 65 | 97 | 95 | 99 | \u003Cbr>98 | 75 | 77 | 88 | 90 | 100 | 100 | 100 | \u003Cbr>100 |  69  | \u003Cbr>97 99 |\n\n\n| Language | Fine-tuned, wit\u003Cbr>mT5 |  |  |  | h frozen encoder\u003Cbr>ByT5 |  |  |  | Few-shot\u003Cbr>PaLM |  |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |\n|  | B | L | XL | XXL | B | L | XL | XXL | 8B | 62B 540B |\n| Arabic | 22 | 60 | 75 | \u003Cbr>87 | 99 | 99 | 100 | \u003Cbr>99 | 32 | \u003Cbr>68 89 |\n| Chinese | 78 | 76 | 83 | 84 | 99 | 98 | 99 | \u003Cbr>99 | 81 | \u003Cbr>93 98 |\n| English | 7 | 32 | 54 | \u003Cbr>71 | 98 | 96 | 99 | \u003Cbr>99 | 71 | \u003Cbr>97 99 |\n| Finnish | 10 | 36 | 62 | \u003Cbr>77 | 98 | 97 | 99 | \u003Cbr>99 | 45 | 84  99  |\n| Korean | 37 | 58 | 77 | 81 | 99 | 99 | 100 | \u003Cbr>99 | 71 | \u003Cbr>88 96 |\n| Russian | 9 | 41 | 57 | \u003Cbr>76 | 99 | 98 | 99 | \u003Cbr>99 | 41 | 86  98  |\n| Thai | 29 | 42 | 46 | \u003Cbr>60 | 99 | 99 | 99 | \u003Cbr>99 | 22 | \u003Cbr>39 63 |\n| Average | 27 | 49 | 65 | \u003Cbr>77 | 99 | 98 | 99 | \u003Cbr>99 | 52 | \u003Cbr>79 92 |\n\n| Error Type | Examples |  |\n| --- | --- | --- |\n| Semantic\u003Cbr>3 | demonstrated ! demonstrafied inquisitiveness ! inquisioness |  |\n| Homophone\u003Cbr>3 |  | accommodate ! accomidate\u003Cbr>Toronto ! Torondo |\n| Add Glyph\u003Cbr>3 |  | labor ! labort debut ! debust |\n| Drop Glyph\u003Cbr>7 |  | stopping ! stoping experiments ! experimets |\n| Repeat Glyph\u003Cbr>7 |  | possible ! posssible locate ! locaate |\n| Merge Glyphs\u003Cbr>7 |  | |\n| Misshape\u003Cbr>7 |  | |\n| No Text\u003Cbr>7 | |  |","cbCait7lAlhifif5","https://ap.wps.com/l/cbCait7lAlhifif5","pdf",6622626,28,"English","# Evaluation results\n## Frequency buckets and top-percent thresholds\n## Language-wise performance\n## Error type analysis","[{\"question\":\"Which models and training settings are compared in the document?\",\"answer\":\"The document compares fine-tuned T5 variants with ByT5, including settings with a frozen encoder and few-shot prompting.\"},{\"question\":\"How is performance reported across different groups of data?\",\"answer\":\"Performance is shown using bucketed frequency ranges (e.g., Top 1%, 1–10%, 10–20%, 20–30%, Bottom 50%) with corresponding scores for different model variants.\"},{\"question\":\"What kinds of generation errors are analyzed?\",\"answer\":\"The document categorizes errors into types including semantic errors, homophone substitutions, added glyphs, dropped glyphs, repeat glyphs, and other text-shape or missing-text issues.\"}]","ACL Long 2023 - Model Fine-tuning Evaluation | PDF",10]