[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"doc-seo-128858-105":3,"detail-sidebar-cat-0-en-105":81,"doc-detail-128858-en":130},{"code":4,"msg":5,"data":6},0,"ok",{"site_id":7,"language":8,"slug":9,"title":10,"keywords":11,"description":12,"schema_data":13,"social_meta":74,"head_meta":76,"extra_data":78,"updated_unix":80},105,"en","adding-vs-averaging-in-distributed-primal-dual-optimization","Adding vs. Averaging in Distributed Primal-Dual Optimization","","Distributed optimization methods for large-scale machine learning face a communication bottleneck that limits how effectively partial work from multiple machines can be aggregated. The paper introduces COCOA+, a communication-efficient primal-dual framework for distributed optimization. Unlike prior schemes that must conservatively average updates, COCOA+ supports additive combination of local updates at each iteration. The work provides stronger primal-dual convergence rate guarantees for COCOA and COCOA+ and extends theory to non-smooth convex losses, validated by extensive experiments on real-world distributed datasets.",{"@graph":14,"@context":73},[15,34,56],{"@type":16,"itemListElement":17},"BreadcrumbList",[18,23,27,31],{"item":19,"name":20,"@type":21,"position":22},"https://docshare.wps.com","Home","ListItem",1,{"item":24,"name":25,"@type":21,"position":26},"https://docshare.wps.com/document/","Document",2,{"item":28,"name":29,"@type":21,"position":30},"https://docshare.wps.com/document/research-report/","Research & Report",3,{"item":32,"name":10,"@type":21,"position":33},"https://docshare.wps.com/document/adding-vs-averaging-in-distributed-primal-dual-optimization/128858/",4,{"url":32,"name":10,"@type":35,"image":36,"author":41,"headline":10,"publisher":44,"fileFormat":47,"inLanguage":8,"description":12,"dateModified":48,"datePublished":49,"encodingFormat":47,"isAccessibleForFree":50,"interactionStatistic":51},"DigitalDocument",{"url":37,"@type":38,"width":39,"height":40},"https://docshare.wps.com/thumbnails/adding-vs-averaging-in-distributed-primal-dual-optimization/128858.png","ImageObject",300,407,{"name":42,"@type":43},"Aria","Person",{"url":19,"name":45,"@type":46},"DocShare","Organization","application/pdf","2026-09-21","2026-08-06",true,{"@type":52,"interactionType":53,"userInteractionCount":55},"InteractionCounter",{"@type":54},"ViewAction",7,{"@type":57,"mainEntity":58},"FAQPage",[59,65,69],{"name":60,"@type":61,"acceptedAnswer":62},"What communication bottleneck problem does the paper address?","Question",{"text":63,"@type":64},"Distributed machine learning is limited by communication delays, which are much slower than reading local data. Existing methods often require communication comparable to local computation, making aggregation inefficient.","Answer",{"name":66,"@type":61,"acceptedAnswer":67},"How does COCOA+ differ from earlier COCOA schemes?",{"text":68,"@type":64},"COCOA+ allows local updates to be additively combined at each iteration, while earlier methods with convergence guarantees primarily permit conservative averaging.",{"name":70,"@type":61,"acceptedAnswer":71},"What types of loss functions are covered by the convergence analysis?",{"text":72,"@type":64},"The theory is extended to non-smooth convex loss functions, including cases such as Support Vector Machines and non-smooth regression variants, with primal-dual convergence rate guarantees.","https://schema.org",{"og:url":32,"og:type":75,"og:title":10,"og:site_name":45,"og:description":12},"article",{"robots":77,"canonical":32},"index,follow",{"doc_id":79,"site_id":7},128858,1786003987,{"code":4,"msg":82,"data":83},"success",[84,88,92,96,101,106,110,114,119,122,126],{"id":22,"doc_module":4,"doc_module_name":25,"category_name":85,"show_sort_weight":86,"slug":87},"Story & Novel",90,"story-novel",{"id":26,"doc_module":4,"doc_module_name":25,"category_name":89,"show_sort_weight":90,"slug":91},"Literature",80,"literature",{"id":33,"doc_module":4,"doc_module_name":25,"category_name":93,"show_sort_weight":94,"slug":95},"Exam",70,"exam",{"id":97,"doc_module":4,"doc_module_name":25,"category_name":98,"show_sort_weight":99,"slug":100},5,"Comic",60,"comic",{"id":102,"doc_module":4,"doc_module_name":25,"category_name":103,"show_sort_weight":104,"slug":105},6,"Technology",50,"technology",{"id":55,"doc_module":4,"doc_module_name":25,"category_name":107,"show_sort_weight":108,"slug":109},"Healthcare",40,"healthcare",{"id":111,"doc_module":4,"doc_module_name":25,"category_name":29,"show_sort_weight":112,"slug":113},8,30,"research-report",{"id":115,"doc_module":4,"doc_module_name":25,"category_name":116,"show_sort_weight":117,"slug":118},9,"Religion & Spirituality",20,"religion-spirituality",{"id":117,"doc_module":4,"doc_module_name":25,"category_name":120,"show_sort_weight":117,"slug":121},"World Cup","world-cup",{"id":123,"doc_module":4,"doc_module_name":25,"category_name":124,"show_sort_weight":123,"slug":125},10,"Lifestyle","lifestyle",{"id":127,"doc_module":4,"doc_module_name":25,"category_name":128,"show_sort_weight":97,"slug":129},19,"General","general",{"code":4,"msg":82,"data":131},{"doc_id":79,"user_id":132,"nickname":42,"user_avatar":133,"doc_module":4,"category_id":111,"category_name":29,"doc_title":10,"doc_description":12,"doc_content":134,"file_id":135,"file_url":136,"file_type":137,"file_size":138,"view_count":55,"is_deleted":4,"is_public":22,"is_downloadable":22,"audit_status":22,"page_count":127,"language":139,"language_code":8,"site_id":7,"html_lang":8,"table_of_contents":140,"faqs":141,"seo_title":142,"seo_description":12,"update_tm":80,"read_time":143},2336474459895,"https://ap-avatar.wpscdn.com/avatar/22000baeef7a5ed0655?x-image-process=image/resize,m_fixed,w_180,h_180&k=1786071322749376916","Adding vs. Averaging in Distributed Primal-Dual Optimization  \nChenxin Ma􀀃  \nIndustrial and Systems Engineering, Lehigh University, USA Virginia Smith􀀃  \nUniversity of California, Berkeley, USA  \nMartin Jaggi  \nETH Z¨urich, Switzerland  \nMichael I. Jordan  \nUniversity of California, Berkeley, USA  \nPeter Richtrik  \nSchool of Mathematics, University of Edinburgh, UK  \nMartin Tak  \nIndustrial and Systems Engineering, Lehigh University, USA  \nCHM 514@LEHIGH . EDUVSMITH @BERKELEY. EDU JAGGI @INF. ETHZ . CH JORDAN @CS . BERKELEY. EDU  \nPETER . RICHTARIK @ED . AC . UK  \nTAKAC . MT@GMAIL . COM  \n􀀃 Authors contributed equally.  \nAbstract  \nDistributed optimization methods for large-scale machine learning suffer from a communication bottleneck. It is difﬁcult to reduce this bottleneck while still efﬁciently and accurately aggregating partial work from different machines. In this paper, we present a novel generalization of the recent communication-efﬁcient primal-dual framework (COCOA) for distributed optimization. Our framework, COCOA+, allows for additive combination of local updates to the global parameters at each iteration, whereas previous schemes with convergence guarantees only allow conservative averaging. We give stronger (primal-dual) convergence rate guarantees for both COCOA as well as our new variants, and generalize the theory for both methods to cover non-smooth convex loss functions. We provide an extensive experimental comparison that shows the markedly improved performance of COCOA+ on several real-world distributed datasets, especially when scaling up the number of machines.  \nProceedings of the 32 nd International Conference on Machine Learning, Lille, France, 2015 . JMLR: W&CP volume 37 . Copyright 2015 by the author(s) .  \n1. Introduction  \nWith the wide availability of large datasets that exceed the storage capacity of single machines, distributed optimization methods for machine learning have become increasingly important. Existing methods require signiﬁcant communication between workers, frequently equaling the amount of local computation (or reading of local data) . Asa result, distributed machine learning suffers signiﬁcantly from a communication bottleneck on real world systems, where communication is typically several orders of magnitudes slower than reading data from main memory.  \nIn this work we focus on optimization problems with empirical loss minimization structure, i.e., objectives that area sum of the loss functions of each datapoint. This includes the most commonly used regularized variants of linear regression and classiﬁcation methods. For this class of problems, the recently proposed COCOA approach (Yang, 2013 ; Jaggi et al., 2014) develops a communicationefﬁcient primal-dual scheme that targets the communication bottleneck, allowing more computation on data-local subproblems native to each machine before communication. By appropriately choosing the amount of local computation per round, this framework allows one to control the trade-off between communication and local computation based on the systems hardware at hand.  \nHowever, the performance of COCOA (as well as related primal SGD-based methods) is signiﬁcantly reduced by the  \nneed to average updates between all machines. As the number of machines K grows, the updates get diluted and slowed by 1=K, e.g., in the case where all machines except one would have already reached the solutions of their respective partial optimization tasks. On the other hand, if the updates are instead added, the algorithms can diverge, as we will observe in the practical experiments below.  \nTo address both described issues, in this paper we develop a novel generalization of the local COCOA subproblems assigned to each worker, making the framework more powerful in the following sense: Without extra computational cost, the set of locally computed updates from the modiﬁed subproblems (one from each machine) can be combined more efﬁciently between machines. The propo","cbCaisl9dAGVtFLS","https://ap.wps.com/l/cbCaisl9dAGVtFLS","pdf",897088,"English","# Introduction\n## Contributions","[{\"question\":\"What communication bottleneck problem does the paper address?\",\"answer\":\"Distributed machine learning is limited by communication delays, which are much slower than reading local data. Existing methods often require communication comparable to local computation, making aggregation inefficient.\"},{\"question\":\"How does COCOA+ differ from earlier COCOA schemes?\",\"answer\":\"COCOA+ allows local updates to be additively combined at each iteration, while earlier methods with convergence guarantees primarily permit conservative averaging.\"},{\"question\":\"What types of loss functions are covered by the convergence analysis?\",\"answer\":\"The theory is extended to non-smooth convex loss functions, including cases such as Support Vector Machines and non-smooth regression variants, with primal-dual convergence rate guarantees.\"}]","Adding vs. Averaging in Distributed Primal-Dual Optimization | PDF",48]