[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"detail-sidebar-cat-0-en-105":3,"doc-seo-140836-105":59,"doc-detail-140836-en":130},{"code":4,"msg":5,"data":6},0,"success",[7,13,18,23,28,33,38,43,48,51,55],{"id":8,"doc_module":4,"doc_module_name":9,"category_name":10,"show_sort_weight":11,"slug":12},1,"Document","Story & Novel",90,"story-novel",{"id":14,"doc_module":4,"doc_module_name":9,"category_name":15,"show_sort_weight":16,"slug":17},2,"Literature",80,"literature",{"id":19,"doc_module":4,"doc_module_name":9,"category_name":20,"show_sort_weight":21,"slug":22},4,"Exam",70,"exam",{"id":24,"doc_module":4,"doc_module_name":9,"category_name":25,"show_sort_weight":26,"slug":27},5,"Comic",60,"comic",{"id":29,"doc_module":4,"doc_module_name":9,"category_name":30,"show_sort_weight":31,"slug":32},6,"Technology",50,"technology",{"id":34,"doc_module":4,"doc_module_name":9,"category_name":35,"show_sort_weight":36,"slug":37},7,"Healthcare",40,"healthcare",{"id":39,"doc_module":4,"doc_module_name":9,"category_name":40,"show_sort_weight":41,"slug":42},8,"Research & Report",30,"research-report",{"id":44,"doc_module":4,"doc_module_name":9,"category_name":45,"show_sort_weight":46,"slug":47},9,"Religion & Spirituality",20,"religion-spirituality",{"id":46,"doc_module":4,"doc_module_name":9,"category_name":49,"show_sort_weight":46,"slug":50},"World Cup","world-cup",{"id":52,"doc_module":4,"doc_module_name":9,"category_name":53,"show_sort_weight":52,"slug":54},10,"Lifestyle","lifestyle",{"id":56,"doc_module":4,"doc_module_name":9,"category_name":57,"show_sort_weight":24,"slug":58},19,"General","general",{"code":4,"msg":60,"data":61},"ok",{"site_id":62,"language":63,"slug":64,"title":65,"keywords":66,"description":67,"schema_data":68,"social_meta":123,"head_meta":125,"extra_data":127,"updated_unix":129},105,"en","m6-multi-modality-to-multi-modality-multitask-mega-transformer-for-unified-pretraining","M6: Multi-Modality-to-Multi-Modality Multitask Mega-transformer for Unified Pretraining","","Multimodal pretraining has proven effective for cross-modal representation learning, yet existing efforts are constrained mainly to English data and lack large-scale Chinese resources. This work introduces the largest Chinese pretraining dataset, M6-Corpus, with over 1.9× images and 292× texts spanning domains such as encyclopedias, question answering, and forum discussions. A unified model, M6, performs multi-modal-to-text and transfer tasks via multitask mega-transformer pretraining. Scaling to 10B parameters achieves strong gains on downstream tasks and shows strong zero-shot potential.",{"@graph":69,"@context":122},[70,84,105],{"@type":71,"itemListElement":72},"BreadcrumbList",[73,77,79,82],{"item":74,"name":75,"@type":76,"position":8},"https://docshare.wps.com","Home","ListItem",{"item":78,"name":9,"@type":76,"position":14},"https://docshare.wps.com/document/",{"item":80,"name":40,"@type":76,"position":81},"https://docshare.wps.com/document/research-report/",3,{"item":83,"name":65,"@type":76,"position":19},"https://docshare.wps.com/document/m6-multi-modality-to-multi-modality-multitask-mega-transformer-for-unified-pretraining/140836/",{"url":83,"name":65,"@type":85,"image":86,"author":91,"headline":65,"publisher":94,"fileFormat":97,"inLanguage":63,"description":67,"dateModified":98,"datePublished":99,"encodingFormat":97,"isAccessibleForFree":100,"interactionStatistic":101},"DigitalDocument",{"url":87,"@type":88,"width":89,"height":90},"https://docshare.wps.com/thumbnails/m6-multi-modality-to-multi-modality-multitask-mega-transformer-for-unified-pretraining/140836.png","ImageObject",300,407,{"name":92,"@type":93},"Ethan Miller","Person",{"url":74,"name":95,"@type":96},"DocShare","Organization","application/pdf","2026-09-16","2026-08-25",true,{"@type":102,"interactionType":103,"userInteractionCount":39},"InteractionCounter",{"@type":104},"ViewAction",{"@type":106,"mainEntity":107},"FAQPage",[108,114,118],{"name":109,"@type":110,"acceptedAnswer":111},"What is the main contribution of this work for Chinese multimodal pretraining?","Question",{"text":112,"@type":113},"It proposes M6-Corpus, a large-scale Chinese dataset for multimodal pretraining, along with a unified model (M6) for multimodal understanding and generation tasks.","Answer",{"name":115,"@type":110,"acceptedAnswer":116},"How does M6-Corpus cover data across domains?",{"text":117,"@type":113},"It is collected from webpages and includes multiple types of content, covering areas such as encyclopedias, question answering, forum discussions, and product descriptions.",{"name":119,"@type":110,"acceptedAnswer":120},"What tasks does the M6 model use during pretraining?",{"text":121,"@type":113},"The model is pretrained with tasks that support text-to-text transfer, image-to-text transfer, and multimodality-to-text transfer, enabling both single-modal and multimodal capabilities.","https://schema.org",{"og:url":83,"og:type":124,"og:title":65,"og:site_name":95,"og:description":67},"article",{"robots":126,"canonical":83},"index,follow",{"doc_id":128,"site_id":62},140836,1787644701,{"code":4,"msg":5,"data":131},{"doc_id":128,"user_id":132,"nickname":92,"user_avatar":133,"doc_module":4,"category_id":39,"category_name":40,"doc_title":65,"doc_description":67,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":39,"is_deleted":4,"is_public":8,"is_downloadable":8,"audit_status":8,"page_count":139,"language":140,"language_code":63,"site_id":62,"html_lang":63,"table_of_contents":141,"faqs":142,"seo_title":143,"seo_description":67,"update_tm":129,"read_time":144},687207017582,"https://ap-avatar.wpscdn.com/davatar_994ba38a5ba835b3df7d355c54d3ed8d","M6: Multi-Modality-to-Multi-Modality Multitask Mega-transformer for Unified Pretraining  \nJunyang Lin 1∗, Rui Men 1∗, An Yang 1∗, Chang Zhou 1 , Yichang Zhang 1 , Peng Wang 1 , Jingren Zhou 1 , Jie Tang2†, Hongxia Yang 1†  \n1DAMO Academy, Alibaba Group  \n2Tsinghua University  \n{junyang.ljy,[menrui.mr](menrui.mr), ya235025,ericzhou.zc,yichang.zyc,zheluo.wp,jingren.zhou,[yang.yhx}@alibaba-inc.com](yang.yhx}@alibaba-inc.com)  \n[jietang@tsinghua.edu.cn](jietang@tsinghua.edu.cn)  \nABSTRACT  \nMultimodal pretraining has demonstrated success in the downstream tasks of cross-modal representation learning. However, it is limited to the English data, and there is still a lack of large-scale dataset for multimodal pretraining in Chinese. In this work, we propose the largest dataset for pretraining in Chinese, which consists of over 1. 9􀀩 􀀗 images and 292􀀜􀀗 texts. The dataset has large coverage over domains, including encyclopedia, question answering, forum discussion, etc. Besides, we propose a method called M6, referring to Multi-Modality-to-Multi-Modality Multitask Mega-transformer, for unified pretraining on the data of single modality and multiple modalities. The model is pretrained with our proposed tasks, including text-to-text transfer, image-to-text transfer, as well as multimodality-to-text transfer. The tasks endow the model with strong capability of understanding and generation. We scale the model to 10 billion parameters, and build the largest pretrained model in Chinese. Experimental results show that our proposed M6 outperforms the baseline in a number of downstream tasks concerning both single modality and multiple modalities, and the 10􀀗 -parameter pretrained model demonstrates strong potential in the setting of zero-shot learning.  \nCCS CONCEPTS  \n• Computing methodologies → Natural language processing; Computer vision.  \nKEYWORDS  \nMulti-modal pretraining; Large-scale pretraining; Cross-modal understanding and generation  \nACM Reference Format:  \nJunyang Lin1∗ , Rui Men1∗ , An Yang1∗ , Chang Zhou1 , Yichang Zhang1 , Peng Wang1 ,, Jingren Zhou1 , Jie Tang2†, Hongxia Yang1†. 2021. M6: MultiModality-to-Multi-Modality Multitask Mega-transformer for Unified Pretraining. In Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD’21), August 14–18, 2021, Virtual Event,  \nPermission to make digital or hard copies of all or part of this work for personal or classroom use is granted without fee provided that copies are not made or distributed for profit or commercial advantage and that copies bear this notice and the full citation on the first page. Copyrights for components of this work owned by others than ACM must be honored. Abstracting with credit is permitted. To copy otherwise, or republish, to post on servers or to redistribute to lists, requires prior specific permission [and/or a fee. Request permissions from permissions@acm.org](and/or a fee. Request permissions from permissions@acm.org).  \nKDD’21, August 14–18, 2021, Virtual Event, Singapore © 2021 Association for Computing Machinery.  \nACM ISBN 978-1-4503-8332-5/21/08…$15.00 [https://doi.org/10.1145/3447548.3467206](https://doi.org/10.1145/3447548.3467206)  \nSingapore. ACM, New York, NY, USA, 11 pages. [https://doi.org/10.1145/](https://doi.org/10.1145/)[ ](https://doi.org/10.1145/)3447548.3467206  \n1 INTRODUCTION  \nPretraining has recently greatly promoted the development of natural language processing (NLP) . A series of studies in pretraining [1, 2, 8, 15, 17, 18, 24, 29, 35, 41, 46] have gradually pushed the limit of model performance in natural language understanding and even natural language generation. Besides, pretraining leveraging large-scale data enables the building of extremely large models with large capacity. The recent GPT-3 with over 175 billion parameters demonstrates that large models can achieve outstanding performance even in the setting of few-shot or zero-shot learning. The rapid development of pretraining in NL","cbCaikMKXuDYE5Um","https://ap.wps.com/l/cbCaikMKXuDYE5Um","pdf",2829181,11,"English","# Abstract\n# Introduction\n## Challenges in Multimodal Pretraining for Chinese\n## M6-Corpus: Large-Scale Chinese Multimodal Dataset\n## M6: Unified Multi-Modal Mega-transformer","[{\"question\":\"What is the main contribution of this work for Chinese multimodal pretraining?\",\"answer\":\"It proposes M6-Corpus, a large-scale Chinese dataset for multimodal pretraining, along with a unified model (M6) for multimodal understanding and generation tasks.\"},{\"question\":\"How does M6-Corpus cover data across domains?\",\"answer\":\"It is collected from webpages and includes multiple types of content, covering areas such as encyclopedias, question answering, forum discussions, and product descriptions.\"},{\"question\":\"What tasks does the M6 model use during pretraining?\",\"answer\":\"The model is pretrained with tasks that support text-to-text transfer, image-to-text transfer, and multimodality-to-text transfer, enabling both single-modal and multimodal capabilities.\"}]","M6: Multi-Modality-to-Multi-Modality Multitask Mega-transformer for Unified Pretraining | PDF",28]