[{"oa_version":"None","corr_author":"1","publication_identifier":{"eissn":["2575-8411"],"issn":["1063-6927"],"isbn":["9798350386059"]},"publication_status":"published","author":[{"last_name":"Chatterjee","full_name":"Chatterjee, Bapi","orcid":"0000-0002-2742-4028","first_name":"Bapi","id":"3C41A08A-F248-11E8-B48F-1D18A9856A87"},{"last_name":"Kungurtsev","full_name":"Kungurtsev, Vyacheslav","first_name":"Vyacheslav"},{"first_name":"Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","orcid":"0000-0003-3650-940X"}],"scopus_import":"1","publisher":"IEEE","date_updated":"2025-09-08T09:23:48Z","year":"2024","external_id":{"isi":["001304430200075"]},"date_published":"2024-07-26T00:00:00Z","abstract":[{"text":"Parallel SGD in a shared-memory setting is oft-represented by the popular Hogwild! algorithm, in which lock-free updates are asynchronously performed by multiple computing processes. Unfortunately, scaling Hogwild! to distributed workers is largely unexplored. Specifically, it is unknown if any adaptation of Hogwild! to the popular decentralized multi-GPU setting offers any competitive speedup, either empirically or theoretically. In this work, we investigate the potential of decentralizing Hogwild! by incorporating simultaneously (a) asynchronous local gradient updates on the shared memory of GPUs, and (b) non-blocking asynchronous decentralized federated averaging. A naive direct implementation shows degradation in performance, arising from scheduling overheads and concurrent write conflicts on GPUs. To mitigate these drawbacks, we investigate and propose a new method, based on careful block selection rules, which update only portions of the parameter vectors. Our experiments show that the resulting decentralized training method exhibits improved throughput and competitive accuracy for standard image classification benchmarks on the CIFAR-10, CIFAR-100, and Imagenet datasets. On the theoretical side, we prove that our method guarantees sublinear ergodic convergence rates for non-convex objectives.","lang":"eng"}],"department":[{"_id":"DaAl"}],"fulldoi":"https://doi.org/10.1109/ICDCS60910.2024.00084","language":[{"iso":"eng"}],"citation":{"ama":"Chatterjee B, Kungurtsev V, Alistarh D-A. Federated SGD with local asynchrony. In: <i>Proceedings of the 44th International Conference on Distributed Computing Systems</i>. IEEE; 2024:857-868. doi:<a href=\"https://doi.org/10.1109/ICDCS60910.2024.00084\">10.1109/ICDCS60910.2024.00084</a>","ista":"Chatterjee B, Kungurtsev V, Alistarh D-A. 2024. Federated SGD with local asynchrony. Proceedings of the 44th International Conference on Distributed Computing Systems. ICDCS: International Conference on Distributed Computing Systems, 857–868.","chicago":"Chatterjee, Bapi, Vyacheslav Kungurtsev, and Dan-Adrian Alistarh. “Federated SGD with Local Asynchrony.” In <i>Proceedings of the 44th International Conference on Distributed Computing Systems</i>, 857–68. IEEE, 2024. <a href=\"https://doi.org/10.1109/ICDCS60910.2024.00084\">https://doi.org/10.1109/ICDCS60910.2024.00084</a>.","short":"B. Chatterjee, V. Kungurtsev, D.-A. Alistarh, in:, Proceedings of the 44th International Conference on Distributed Computing Systems, IEEE, 2024, pp. 857–868.","apa":"Chatterjee, B., Kungurtsev, V., &#38; Alistarh, D.-A. (2024). Federated SGD with local asynchrony. In <i>Proceedings of the 44th International Conference on Distributed Computing Systems</i> (pp. 857–868). Jersey City, NJ, United States: IEEE. <a href=\"https://doi.org/10.1109/ICDCS60910.2024.00084\">https://doi.org/10.1109/ICDCS60910.2024.00084</a>","ieee":"B. Chatterjee, V. Kungurtsev, and D.-A. Alistarh, “Federated SGD with local asynchrony,” in <i>Proceedings of the 44th International Conference on Distributed Computing Systems</i>, Jersey City, NJ, United States, 2024, pp. 857–868.","mla":"Chatterjee, Bapi, et al. “Federated SGD with Local Asynchrony.” <i>Proceedings of the 44th International Conference on Distributed Computing Systems</i>, IEEE, 2024, pp. 857–68, doi:<a href=\"https://doi.org/10.1109/ICDCS60910.2024.00084\">10.1109/ICDCS60910.2024.00084</a>."},"status":"public","month":"07","type":"conference","doi":"10.1109/ICDCS60910.2024.00084","page":"857-868","article_processing_charge":"No","isi":1,"user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","title":"Federated SGD with local asynchrony","_id":"18070","publication":"Proceedings of the 44th International Conference on Distributed Computing Systems","date_created":"2024-09-15T22:01:41Z","conference":{"start_date":"2024-07-23","name":"ICDCS: International Conference on Distributed Computing Systems","end_date":"2024-07-26","location":"Jersey City, NJ, United States"},"quality_controlled":"1","day":"26"},{"isi":1,"article_processing_charge":"No","page":"1377-1387","doi":"10.1109/ICDCS60910.2024.00129","type":"conference","citation":{"apa":"Tsimos, G., Kichidis, A., Sonnino, A., &#38; Kokoris Kogias, E. (2024). HammerHead: Leader reputation for dynamic scheduling. In <i>Proceedings - International Conference on Distributed Computing Systems</i> (pp. 1377–1387). Jersey City, NJ, United States: IEEE. <a href=\"https://doi.org/10.1109/ICDCS60910.2024.00129\">https://doi.org/10.1109/ICDCS60910.2024.00129</a>","ieee":"G. Tsimos, A. Kichidis, A. Sonnino, and E. Kokoris Kogias, “HammerHead: Leader reputation for dynamic scheduling,” in <i>Proceedings - International Conference on Distributed Computing Systems</i>, Jersey City, NJ, United States, 2024, pp. 1377–1387.","mla":"Tsimos, Giorgos, et al. “HammerHead: Leader Reputation for Dynamic Scheduling.” <i>Proceedings - International Conference on Distributed Computing Systems</i>, IEEE, 2024, pp. 1377–87, doi:<a href=\"https://doi.org/10.1109/ICDCS60910.2024.00129\">10.1109/ICDCS60910.2024.00129</a>.","ama":"Tsimos G, Kichidis A, Sonnino A, Kokoris Kogias E. HammerHead: Leader reputation for dynamic scheduling. In: <i>Proceedings - International Conference on Distributed Computing Systems</i>. IEEE; 2024:1377-1387. doi:<a href=\"https://doi.org/10.1109/ICDCS60910.2024.00129\">10.1109/ICDCS60910.2024.00129</a>","ista":"Tsimos G, Kichidis A, Sonnino A, Kokoris Kogias E. 2024. HammerHead: Leader reputation for dynamic scheduling. Proceedings - International Conference on Distributed Computing Systems. ICDCS: International Conference on Distributed Computing Systems, 1377–1387.","chicago":"Tsimos, Giorgos, Anastasios Kichidis, Alberto Sonnino, and Eleftherios Kokoris Kogias. “HammerHead: Leader Reputation for Dynamic Scheduling.” In <i>Proceedings - International Conference on Distributed Computing Systems</i>, 1377–87. IEEE, 2024. <a href=\"https://doi.org/10.1109/ICDCS60910.2024.00129\">https://doi.org/10.1109/ICDCS60910.2024.00129</a>.","short":"G. Tsimos, A. Kichidis, A. Sonnino, E. Kokoris Kogias, in:, Proceedings - International Conference on Distributed Computing Systems, IEEE, 2024, pp. 1377–1387."},"month":"07","status":"public","conference":{"start_date":"2024-07-23","name":"ICDCS: International Conference on Distributed Computing Systems","end_date":"2024-07-26","location":"Jersey City, NJ, United States"},"quality_controlled":"1","day":"26","oa":1,"date_created":"2024-09-15T22:01:41Z","publication":"Proceedings - International Conference on Distributed Computing Systems","_id":"18071","title":"HammerHead: Leader reputation for dynamic scheduling","user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","main_file_link":[{"open_access":"1","url":"https://arxiv.org/abs/2309.12713"}],"date_updated":"2025-09-08T09:42:36Z","publisher":"IEEE","year":"2024","publication_status":"published","scopus_import":"1","author":[{"first_name":"Giorgos","last_name":"Tsimos","full_name":"Tsimos, Giorgos"},{"full_name":"Kichidis, Anastasios","last_name":"Kichidis","first_name":"Anastasios"},{"first_name":"Alberto","last_name":"Sonnino","full_name":"Sonnino, Alberto"},{"last_name":"Kokoris Kogias","full_name":"Kokoris Kogias, Eleftherios","id":"f5983044-d7ef-11ea-ac6d-fd1430a26d30","first_name":"Eleftherios"}],"publication_identifier":{"isbn":["9798350386059"],"eissn":["2575-8411"],"issn":["1063-6927"]},"oa_version":"Preprint","fulldoi":"https://doi.org/10.1109/ICDCS60910.2024.00129","language":[{"iso":"eng"}],"acknowledgement":"This work is supported by Mysten Labs. We thank the Mysten Labs Engineering teams for valuable feedback broadly, and specifically to Laura Makdah for helping implementing the early reputation score system for validators and Dmitry Perelman for managing the overall implementation effort.","arxiv":1,"date_published":"2024-07-26T00:00:00Z","department":[{"_id":"ElKo"}],"abstract":[{"text":"Recent advancements on DAG-based consensus protocols allow for blockchains with improved metrics and properties, such as throughput and censorship-resistance. Variants of the Bullshark [18] consensus protocol are adopted for practical use by the Sui blockchain, for improved latency. However, the protocol is leader-based, and is strongly affected by crashed leaders that can lead to various performance issues, for example, decreased transaction throughput. In this paper, we propose HammerHead, a DAG-based consensus protocol, that is inspired by Carousel [8] and provides Leader-Utilization. Our proposal differs from Carousel, which is built for a chained consensus protocol; in HammerHead chain quality is inherited by the DAG. HammerHead needs to preserve safety and liveness, despite validators committing leader vertices asynchronously. The key idea is to update leader schedules dynamically, based on the validators' scores during the previous schedule. We implement HammerHead and show a minor improvement in performance for cases without faults. The major improvements in comparison to Bullshark appear in faulty settings. Specifically, we show a drastic, 2x-latency improvement and up to 40% increased throughput when crash faults occur (100 validators, 33 faults).","lang":"eng"}],"external_id":{"arxiv":["2309.12713"],"isi":["001304430200120"]}},{"external_id":{"arxiv":["1405.5461"]},"arxiv":1,"language":[{"iso":"eng"}],"fulldoi":"https://doi.org/10.1109/ICDCS.2014.43","abstract":[{"text":"The long-lived renaming problem appears in shared-memory systems where a set of threads need to register and deregister frequently from the computation, while concurrent operations scan the set of currently registered threads. Instances of this problem show up in concurrent implementations of transactional memory, flat combining, thread barriers, and memory reclamation schemes for lock-free data structures. In this paper, we analyze a randomized solution for long-lived renaming. The algorithmic technique we consider, called the Level Array, has previously been used for hashing and one-shot (single-use) renaming. Our main contribution is to prove that, in long-lived executions, where processes may register and deregister polynomially many times, the technique guarantees constant steps on average and O (log log n) steps with high probability for registering, unit cost for deregistering, and O (n) steps for collect queries, where n is an upper bound on the number of processes that may be active at any point in time. We also show that the algorithm has the surprising property that it is self-healing: under reasonable assumptions on the schedule, operations running while the data structure is in a degraded state implicitly help the data structure re-balance itself. This subtle mechanism obviates the need for expensive periodic rebuilding procedures. Our benchmarks validate this approach, showing that, for typical use parameters, the average number of steps a process takes to register is less than two and the worst-case number of steps is bounded by six, even in executions with billions of operations. We contrast this with other randomized implementations, whose worst-case behavior we show to be unreliable, and with deterministic implementations, whose cost is linear in n.","lang":"eng"}],"extern":"1","date_published":"2014-08-29T00:00:00Z","publication_identifier":{"issn":["1063-6927"],"eisbn":["9781479951697"]},"oa_version":"Preprint","year":"2014","date_updated":"2026-09-09T12:34:00Z","publisher":"IEEE","OA_place":"repository","author":[{"orcid":"0000-0003-3650-940X","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","first_name":"Dan-Adrian"},{"first_name":"Justin","last_name":"Kopinsky","full_name":"Kopinsky, Justin"},{"first_name":"Alexander","full_name":"Matveev, Alexander","last_name":"Matveev"},{"first_name":"Nir","full_name":"Shavit, Nir","last_name":"Shavit"}],"publication_status":"published","title":"The levelarray: A fast, practical long-lived renaming algorithm","main_file_link":[{"open_access":"1","url":"https://arxiv.org/abs/1405.5461"}],"user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","day":"29","oa":1,"conference":{"start_date":"2014-06-30","name":"ICDCS: International Conference on Distributed Computing Systems","end_date":"2014-07-03","location":"Madrid, Spain"},"date_created":"2018-12-11T11:48:26Z","_id":"775","publication":"Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems","type":"conference","month":"08","status":"public","citation":{"ieee":"D.-A. Alistarh, J. Kopinsky, A. Matveev, and N. Shavit, “The levelarray: A fast, practical long-lived renaming algorithm,” in <i>Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems</i>, Madrid, Spain, 2014, pp. 348–357.","mla":"Alistarh, Dan-Adrian, et al. “The Levelarray: A Fast, Practical Long-Lived Renaming Algorithm.” <i>Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems</i>, IEEE, 2014, pp. 348–57, doi:<a href=\"https://doi.org/10.1109/ICDCS.2014.43\">10.1109/ICDCS.2014.43</a>.","apa":"Alistarh, D.-A., Kopinsky, J., Matveev, A., &#38; Shavit, N. (2014). The levelarray: A fast, practical long-lived renaming algorithm. In <i>Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems</i> (pp. 348–357). Madrid, Spain: IEEE. <a href=\"https://doi.org/10.1109/ICDCS.2014.43\">https://doi.org/10.1109/ICDCS.2014.43</a>","chicago":"Alistarh, Dan-Adrian, Justin Kopinsky, Alexander Matveev, and Nir Shavit. “The Levelarray: A Fast, Practical Long-Lived Renaming Algorithm.” In <i>Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems</i>, 348–57. IEEE, 2014. <a href=\"https://doi.org/10.1109/ICDCS.2014.43\">https://doi.org/10.1109/ICDCS.2014.43</a>.","short":"D.-A. Alistarh, J. Kopinsky, A. Matveev, N. Shavit, in:, Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems, IEEE, 2014, pp. 348–357.","ama":"Alistarh D-A, Kopinsky J, Matveev A, Shavit N. The levelarray: A fast, practical long-lived renaming algorithm. In: <i>Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems</i>. IEEE; 2014:348-357. doi:<a href=\"https://doi.org/10.1109/ICDCS.2014.43\">10.1109/ICDCS.2014.43</a>","ista":"Alistarh D-A, Kopinsky J, Matveev A, Shavit N. 2014. The levelarray: A fast, practical long-lived renaming algorithm. Proceedings of the 2014 IEEE 34th International Conference on Distributed Computing Systems. ICDCS: International Conference on Distributed Computing Systems, 348–357."},"OA_type":"green","article_processing_charge":"No","publist_id":"6883","page":"348 - 357","doi":"10.1109/ICDCS.2014.43"}]
