[{"main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2310.06927"}],"doi":"10.1007/978-3-031-85747-8_6","abstract":[{"text":"We investigate the problem of accurate sparse fine-tuning of large language models (LLMs), that is, fine-tuning pre-trained LLMs on specialized tasks, while inducing sparsity in their weights. Our work is motivated by experiments showing that standard loss-based fine-tuning methods are not able to achieve high accuracy in this setting, especially at high sparsity targets. To address this issue, we perform a detailed study of knowledge distillation losses for fine-tuning of sparse models. We determine an L2-based distillation approach that we term ‘SquareHead’, which enables accurate recovery even at higher sparsities. Investigating the question of efficient inference, we show that sparse LLMs can be executed faster by taking advantage of sparsity. Specifically, we exhibit end-to-end results showing speedups enabled by sparsity, while recovering accuracy, on the following models and tasks, respectively: T5 for language translation, Whisper for speech translation, and open GPT-type models such as the Mosaic Pre-Trained Transformer (MPT) and Llama-2 models for text generation. In particular, for popular generative tasks, we show for the first time that sparse fine-tuning can reach 75% sparsity without drops in accuracy, and provide notable end-to-end speedups for inference on CPUs. Moreover, we also highlight that sparsity is compatible with other compression approaches, such as quantization.","lang":"eng"}],"year":"2025","alternative_title":["Machine Translation: Technologies and Applications"],"publication_status":"published","OA_type":"green","article_processing_charge":"No","language":[{"iso":"eng"}],"day":"05","citation":{"mla":"Kurtic, Eldar, et al. “Sparse Fine-Tuning for Inference Acceleration of Large Language Models.” <i>Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques</i>, edited by Peyman Passban et al., Springer Nature, 2025, pp. 83–97, doi:<a href=\"https://doi.org/10.1007/978-3-031-85747-8_6\">10.1007/978-3-031-85747-8_6</a>.","ama":"Kurtic E, Kuznedelev D, Frantar E, et al. Sparse Fine-Tuning for Inference Acceleration of Large Language Models. In: Passban P, Way A, Rezagholizadeh M, eds. <i>Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques</i>. Springer Nature; 2025:83-97. doi:<a href=\"https://doi.org/10.1007/978-3-031-85747-8_6\">10.1007/978-3-031-85747-8_6</a>","ieee":"E. Kurtic <i>et al.</i>, “Sparse Fine-Tuning for Inference Acceleration of Large Language Models,” in <i>Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques</i>, P. Passban, A. Way, and M. Rezagholizadeh, Eds. Springer Nature, 2025, pp. 83–97.","chicago":"Kurtic, Eldar, Denis Kuznedelev, Elias Frantar, Michael Goinv, Shubhra Pandit, Abhinav Agarwalla, Tuan Nguyen, Alexandre Marques, Mark Kurtz, and Dan-Adrian Alistarh. “Sparse Fine-Tuning for Inference Acceleration of Large Language Models.” In <i>Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques</i>, edited by Peyman Passban, Andy Way, and Mehdi Rezagholizadeh, 83–97. Springer Nature, 2025. <a href=\"https://doi.org/10.1007/978-3-031-85747-8_6\">https://doi.org/10.1007/978-3-031-85747-8_6</a>.","ista":"Kurtic E, Kuznedelev D, Frantar E, Goinv M, Pandit S, Agarwalla A, Nguyen T, Marques A, Kurtz M, Alistarh D-A. 2025.Sparse Fine-Tuning for Inference Acceleration of Large Language Models. In: Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques. Machine Translation: Technologies and Applications, , 83–97.","short":"E. Kurtic, D. Kuznedelev, E. Frantar, M. Goinv, S. Pandit, A. Agarwalla, T. Nguyen, A. Marques, M. Kurtz, D.-A. Alistarh, in:, P. Passban, A. Way, M. Rezagholizadeh (Eds.), Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques, Springer Nature, 2025, pp. 83–97.","apa":"Kurtic, E., Kuznedelev, D., Frantar, E., Goinv, M., Pandit, S., Agarwalla, A., … Alistarh, D.-A. (2025). Sparse Fine-Tuning for Inference Acceleration of Large Language Models. In P. Passban, A. Way, &#38; M. Rezagholizadeh (Eds.), <i>Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques</i> (pp. 83–97). Springer Nature. <a href=\"https://doi.org/10.1007/978-3-031-85747-8_6\">https://doi.org/10.1007/978-3-031-85747-8_6</a>"},"date_updated":"2026-02-19T09:26:54Z","month":"07","date_published":"2025-07-05T00:00:00Z","oa_version":"Preprint","corr_author":"1","status":"public","department":[{"_id":"DaAl"},{"_id":"GradSch"}],"type":"book_chapter","page":"83-97","publication_identifier":{"eisbn":["9783031857478"],"issn":["2522-8021"],"eissn":["2522-803X"],"isbn":["9783031857461"]},"publication":"Enhancing LLM Performance. Efficacy, Fine-Tuning, and Inference Techniques","_id":"21257","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","publisher":"Springer Nature","title":"Sparse Fine-Tuning for Inference Acceleration of Large Language Models","external_id":{"arxiv":["2310.06927"]},"arxiv":1,"date_created":"2026-02-16T15:57:53Z","oa":1,"OA_place":"repository","author":[{"id":"47beb3a5-07b5-11eb-9b87-b108ec578218","full_name":"Kurtic, Eldar","last_name":"Kurtic","first_name":"Eldar"},{"full_name":"Kuznedelev, Denis","last_name":"Kuznedelev","first_name":"Denis"},{"first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"first_name":"Michael","full_name":"Goinv, Michael","last_name":"Goinv"},{"first_name":"Shubhra","last_name":"Pandit","full_name":"Pandit, Shubhra"},{"first_name":"Abhinav","last_name":"Agarwalla","full_name":"Agarwalla, Abhinav"},{"first_name":"Tuan","full_name":"Nguyen, Tuan","last_name":"Nguyen"},{"full_name":"Marques, Alexandre","last_name":"Marques","first_name":"Alexandre"},{"last_name":"Kurtz","full_name":"Kurtz, Mark","first_name":"Mark"},{"first_name":"Dan-Adrian","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87"}],"editor":[{"last_name":"Passban","full_name":"Passban, Peyman","first_name":"Peyman"},{"full_name":"Way, Andy","last_name":"Way","first_name":"Andy"},{"last_name":"Rezagholizadeh","full_name":"Rezagholizadeh, Mehdi","first_name":"Mehdi"}],"quality_controlled":"1","acknowledgement":"We would like to thank Eugenia Iofinova for useful comments on an earlier version of this draft, and Artur Niederfahrenhorst for useful suggestions regarding fine-tuning on the GSM8k dataset."},{"publisher":"Association for Computing Machinery","conference":{"location":"Las Vegas, NV, United States","end_date":"2025-03-05","start_date":"2025-03-01","name":"PPoPP: Symposium on Principles and Practice of Parallel Programming"},"external_id":{"isi":["001437826500019"],"arxiv":["2408.11743"]},"arxiv":1,"title":"MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models","date_created":"2025-06-23T13:51:58Z","quality_controlled":"1","author":[{"full_name":"Frantar, Elias","last_name":"Frantar","first_name":"Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"full_name":"Castro, Roberto L.","last_name":"Castro","first_name":"Roberto L."},{"first_name":"Jiale","full_name":"Chen, Jiale","last_name":"Chen","orcid":"0000-0001-5337-5875","id":"4d0a9064-1ff6-11ee-9fa6-ec046c604785"},{"full_name":"Hoefler, Torsten","last_name":"Hoefler","first_name":"Torsten"},{"full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X"}],"acknowledgement":"The authors would like to thank the Neural Magic team, in particular Michael Goin, Alexander Matveev, and Rob Shaw, for support with the vLLM integration. This research was supported in part by generous grants from NVIDIA and Google.","OA_place":"publisher","oa":1,"scopus_import":"1","related_material":{"record":[{"relation":"software","id":"19884","status":"public"}]},"corr_author":"1","status":"public","ddc":["000"],"oa_version":"Published Version","department":[{"_id":"DaAl"}],"file_date_updated":"2025-06-24T06:04:17Z","isi":1,"publication_identifier":{"isbn":["9798400714436"]},"page":"239-251","has_accepted_license":"1","type":"conference","user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","_id":"19877","publication":"Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","language":[{"iso":"eng"}],"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"citation":{"apa":"Frantar, E., Castro, R. L., Chen, J., Hoefler, T., &#38; Alistarh, D.-A. (2025). MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models. In <i>Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming</i> (pp. 239–251). Las Vegas, NV, United States: Association for Computing Machinery. <a href=\"https://doi.org/10.1145/3710848.3710871\">https://doi.org/10.1145/3710848.3710871</a>","short":"E. Frantar, R.L. Castro, J. Chen, T. Hoefler, D.-A. Alistarh, in:, Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming, Association for Computing Machinery, 2025, pp. 239–251.","ista":"Frantar E, Castro RL, Chen J, Hoefler T, Alistarh D-A. 2025. MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models. Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming. PPoPP: Symposium on Principles and Practice of Parallel Programming, 239–251.","ieee":"E. Frantar, R. L. Castro, J. Chen, T. Hoefler, and D.-A. Alistarh, “MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models,” in <i>Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming</i>, Las Vegas, NV, United States, 2025, pp. 239–251.","chicago":"Frantar, Elias, Roberto L. Castro, Jiale Chen, Torsten Hoefler, and Dan-Adrian Alistarh. “MARLIN: Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models.” In <i>Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming</i>, 239–51. Association for Computing Machinery, 2025. <a href=\"https://doi.org/10.1145/3710848.3710871\">https://doi.org/10.1145/3710848.3710871</a>.","ama":"Frantar E, Castro RL, Chen J, Hoefler T, Alistarh D-A. MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models. In: <i>Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming</i>. Association for Computing Machinery; 2025:239-251. doi:<a href=\"https://doi.org/10.1145/3710848.3710871\">10.1145/3710848.3710871</a>","mla":"Frantar, Elias, et al. “MARLIN: Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models.” <i>Proceedings of the 30th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming</i>, Association for Computing Machinery, 2025, pp. 239–51, doi:<a href=\"https://doi.org/10.1145/3710848.3710871\">10.1145/3710848.3710871</a>."},"day":"28","date_updated":"2025-09-30T13:41:57Z","file":[{"access_level":"open_access","success":1,"content_type":"application/pdf","relation":"main_file","file_id":"19883","file_size":1330044,"date_updated":"2025-06-24T06:04:17Z","creator":"dernst","date_created":"2025-06-24T06:04:17Z","checksum":"a0566ea3c168e8273501a5eb7d767cf8","file_name":"2025_PPoPP_Frantar.pdf"}],"date_published":"2025-02-28T00:00:00Z","month":"02","doi":"10.1145/3710848.3710871","publication_status":"published","year":"2025","abstract":[{"text":"As inference on Large Language Models (LLMs) emerges as an important workload in machine learning applications, model weight quantization has become a standard technique for efficient GPU deployment. Quantization not only reduces model size, but has also been shown to yield substantial speedups for single-user inference, due to reduced memory movement, with low accuracy impact. Yet, it remains a key open question whether speedups are achievable also in batched settings with multiple parallel clients, which are highly relevant for practical serving. It is unclear whether GPU kernels can be designed to remain practically memory-bound, while supporting the substantially increased compute requirements of batched workloads.\r\nIn this paper, we resolve this question positively by introducing a new design for Mixed-precision Auto-Regressive LINear kernels, called MARLIN. Concretely, given a model whose weights are compressed via quantization to, e.g., 4 bits per element, MARLIN shows that batchsizes up to 16-32 can be practically supported with close to maximum (4×) quantization speedup, and larger batchsizes up to 64-128 with gradually decreasing, but still significant, acceleration. MARLIN accomplishes this via a combination of techniques, such as asynchronous memory access, complex task scheduling and pipelining, and bespoke quantization support. Our experiments show that MARLIN's near-optimal performance on individual LLM layers across different scenarios can also lead to significant end-to-end LLM inference speedups (of up to 2.8×) when integrated with the popular vLLM open-source serving engine. Finally, we show that MARLIN is extensible to further compression techniques, like NVIDIA 2:4 sparsity, leading to additional speedups.","lang":"eng"}],"OA_type":"hybrid","article_processing_charge":"Yes (via OA deal)"},{"_id":"18975","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","publication":"41st International Conference on Machine Learning","volume":235,"publication_identifier":{"eissn":["2640-3498"]},"page":"35910-35933","type":"conference","acknowledged_ssus":[{"_id":"CampIT"}],"intvolume":"       235","department":[{"_id":"DaAl"}],"scopus_import":"1","oa_version":"Preprint","status":"public","corr_author":"1","OA_place":"repository","author":[{"id":"449f7a18-f128-11eb-9611-9b430c0c6333","last_name":"Modoranu","full_name":"Modoranu, Ionut-Vlad","first_name":"Ionut-Vlad"},{"first_name":"Aleksei","full_name":"Kalinov, Aleksei","last_name":"Kalinov","id":"44b7120e-eb97-11eb-a6c2-e1557aa81d02","orcid":"0000-0003-2189-3904"},{"full_name":"Kurtic, Eldar","last_name":"Kurtic","first_name":"Eldar","id":"47beb3a5-07b5-11eb-9b87-b108ec578218"},{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","full_name":"Frantar, Elias","last_name":"Frantar","first_name":"Elias"},{"orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian"}],"acknowledgement":"The authors thank Adrian Vladu, Razvan Pascanu, Alexandra Peste, Mher Safaryan for their valuable feedback, the IT department from Institute of Science and Technology Austria for the hardware support and Weights and Biases for the infrastructure to track all our experiments.","quality_controlled":"1","oa":1,"date_created":"2025-01-30T07:53:22Z","external_id":{"arxiv":["2306.06098"]},"arxiv":1,"title":"Error feedback can accurately compress preconditioners","conference":{"name":"ICML: International Conference on Machine Learning","start_date":"2024-07-21","end_date":"2024-07-27","location":"Vienna, Austria"},"publisher":"ML Research Press","OA_type":"green","article_processing_charge":"No","publication_status":"published","abstract":[{"lang":"eng","text":"Leveraging second-order information about the loss at the scale of deep networks is one of the main lines of approach for improving the performance of current optimizers for deep learning. Yet, existing approaches for accurate full-matrix preconditioning, such as Full-Matrix Adagrad (GGT) or Matrix-Free Approximate Curvature (M-FAC) suffer from massive storage costs when applied even to small-scale models, as they must store a sliding window of gradients, whose memory requirements are multiplicative in the model dimension. In this paper, we address this issue via a novel and efficient error-feedback technique that can be applied to compress preconditioners by up to two orders of magnitude in practice, without loss of convergence. Specifically, our approach compresses the gradient information via sparsification or low-rank compression before it is fed into the preconditioner, feeding the compression error back into future iterations. Extensive experiments on deep neural networks show that this approach can compress full-matrix preconditioners to up to 99% sparsity without accuracy loss, effectively removing the memory overhead of fullmatrix preconditioners such as GGT and M-FAC."}],"alternative_title":["PMLR"],"year":"2024","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2306.06098","open_access":"1"}],"month":"07","date_published":"2024-07-30T00:00:00Z","date_updated":"2025-01-30T07:54:16Z","day":"30","citation":{"mla":"Modoranu, Ionut-Vlad, et al. “Error Feedback Can Accurately Compress Preconditioners.” <i>41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 35910–33.","ama":"Modoranu I-V, Kalinov A, Kurtic E, Frantar E, Alistarh D-A. Error feedback can accurately compress preconditioners. In: <i>41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:35910-35933.","ieee":"I.-V. Modoranu, A. Kalinov, E. Kurtic, E. Frantar, and D.-A. Alistarh, “Error feedback can accurately compress preconditioners,” in <i>41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 35910–35933.","chicago":"Modoranu, Ionut-Vlad, Aleksei Kalinov, Eldar Kurtic, Elias Frantar, and Dan-Adrian Alistarh. “Error Feedback Can Accurately Compress Preconditioners.” In <i>41st International Conference on Machine Learning</i>, 235:35910–33. ML Research Press, 2024.","ista":"Modoranu I-V, Kalinov A, Kurtic E, Frantar E, Alistarh D-A. 2024. Error feedback can accurately compress preconditioners. 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 35910–35933.","short":"I.-V. Modoranu, A. Kalinov, E. Kurtic, E. Frantar, D.-A. Alistarh, in:, 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 35910–35933.","apa":"Modoranu, I.-V., Kalinov, A., Kurtic, E., Frantar, E., &#38; Alistarh, D.-A. (2024). Error feedback can accurately compress preconditioners. In <i>41st International Conference on Machine Learning</i> (Vol. 235, pp. 35910–35933). Vienna, Austria: ML Research Press."},"language":[{"iso":"eng"}]},{"article_processing_charge":"No","OA_type":"green","abstract":[{"text":"Recent advances in large language model (LLM) pretraining have led to high-quality LLMs with impressive abilities. By compressing such LLMs via quantization to 3-4 bits per parameter, they can fit into memory-limited devices such as laptops and mobile phones, enabling personalized use. Quantizing models to 3-4 bits per parameter can lead to moderate to high accuracy losses, especially for smaller models (1-10B parameters), which are suitable for edge deployment. To address this accuracy issue, we introduce the Sparse-Quantized Representation (SpQR), a new compressed format and quantization technique that enables for the first time \\emph{near-lossless} compression of LLMs across model scales while reaching similar compression levels to previous methods. SpQR works by identifying and isolating \\emph{outlier weights}, which cause particularly large quantization errors, and storing them in higher precision while compressing all other weights to 3-4 bits, and achieves relative accuracy losses of less than \r\n in perplexity for highly-accurate LLaMA and Falcon LLMs. This makes it possible to run a 33B parameter LLM on a single 24 GB consumer GPU without performance degradation at 15% speedup, thus making powerful LLMs available to consumers without any downsides. SpQR comes with efficient algorithms for both encoding weights into its format, as well as decoding them efficiently at runtime. Specifically, we provide an efficient GPU inference algorithm for SpQR, which yields faster inference than 16-bit baselines at similar accuracy while enabling memory compression gains of more than 4x.","lang":"eng"}],"year":"2024","publication_status":"published","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2306.03078","open_access":"1"}],"month":"05","date_published":"2024-05-15T00:00:00Z","date_updated":"2025-01-30T08:27:47Z","day":"15","citation":{"chicago":"Dettmers, Tim, Ruslan A. Svirschevski, Vage Egiazarian, Denis Kuznedelev, Elias Frantar, Saleh Ashkboos, Alexander Borzunov, Torsten Hoefler, and Dan-Adrian Alistarh. “SpQR: A Sparse-Quantized Representation for near-Lossless LLM Weight Compression.” In <i>12th International Conference on Learning Representations</i>. OpenReview, 2024.","ieee":"T. Dettmers <i>et al.</i>, “SpQR: A sparse-quantized representation for near-lossless LLM weight compression,” in <i>12th International Conference on Learning Representations</i>, Vienna, Austria, 2024.","ista":"Dettmers T, Svirschevski RA, Egiazarian V, Kuznedelev D, Frantar E, Ashkboos S, Borzunov A, Hoefler T, Alistarh D-A. 2024. SpQR: A sparse-quantized representation for near-lossless LLM weight compression. 12th International Conference on Learning Representations. ICLR: International Conference on Learning Representations.","mla":"Dettmers, Tim, et al. “SpQR: A Sparse-Quantized Representation for near-Lossless LLM Weight Compression.” <i>12th International Conference on Learning Representations</i>, OpenReview, 2024.","ama":"Dettmers T, Svirschevski RA, Egiazarian V, et al. SpQR: A sparse-quantized representation for near-lossless LLM weight compression. In: <i>12th International Conference on Learning Representations</i>. OpenReview; 2024.","apa":"Dettmers, T., Svirschevski, R. A., Egiazarian, V., Kuznedelev, D., Frantar, E., Ashkboos, S., … Alistarh, D.-A. (2024). SpQR: A sparse-quantized representation for near-lossless LLM weight compression. In <i>12th International Conference on Learning Representations</i>. Vienna, Austria: OpenReview.","short":"T. Dettmers, R.A. Svirschevski, V. Egiazarian, D. Kuznedelev, E. Frantar, S. Ashkboos, A. Borzunov, T. Hoefler, D.-A. Alistarh, in:, 12th International Conference on Learning Representations, OpenReview, 2024."},"language":[{"iso":"eng"}],"publication":"12th International Conference on Learning Representations","_id":"18977","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","type":"conference","department":[{"_id":"DaAl"}],"oa_version":"Preprint","status":"public","scopus_import":"1","oa":1,"OA_place":"repository","acknowledgement":"Denis Kuznedelev acknowledges the support from the Russian Ministry of Science and Higher\r\nEducation, grant No. 075-10-2021-068. Ruslan Svirschevski and Vage Egiazarian and Denis\r\nKuznedelev were supported by the grant for research centers in the field of AI provided by the\r\nAnalytical Center for the Government of the Russian Federation (ACRF) in accordance with the\r\nagreement on the provision of subsidies (identifier of the agreement 000000D730321P5Q0002) and the agreement with HSE University No. 70-2021-00139.","quality_controlled":"1","author":[{"first_name":"Tim","last_name":"Dettmers","full_name":"Dettmers, Tim"},{"full_name":"Svirschevski, Ruslan A.","last_name":"Svirschevski","first_name":"Ruslan A."},{"first_name":"Vage","last_name":"Egiazarian","full_name":"Egiazarian, Vage"},{"first_name":"Denis","full_name":"Kuznedelev, Denis","last_name":"Kuznedelev"},{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar"},{"first_name":"Saleh","last_name":"Ashkboos","full_name":"Ashkboos, Saleh"},{"last_name":"Borzunov","full_name":"Borzunov, Alexander","first_name":"Alexander"},{"full_name":"Hoefler, Torsten","last_name":"Hoefler","first_name":"Torsten"},{"last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","first_name":"Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X"}],"date_created":"2025-01-30T08:26:59Z","title":"SpQR: A sparse-quantized representation for near-lossless LLM weight compression","external_id":{"arxiv":["2306.03078"]},"arxiv":1,"conference":{"location":"Vienna, Austria","end_date":"2024-05-11","start_date":"2024-05-07","name":"ICLR: International Conference on Learning Representations"},"publisher":"OpenReview"},{"main_file_link":[{"open_access":"1","url":"https://doi.org/10.5281/ZENODO.14213091"}],"oa_version":"Published Version","status":"public","related_material":{"record":[{"relation":"used_for_analysis_in","status":"public","id":"19877"}]},"corr_author":"1","ddc":["000"],"doi":"10.5281/ZENODO.14213091","department":[{"_id":"DaAl"}],"abstract":[{"text":"This is Marlin, a Mixed Auto-Regressive Linear kernel (and the name of one of the planet's fastest fish), an extremely optimized FP16xINT4 matmul kernel aimed at LLM inference that can deliver close to ideal (4x) speedups up to batchsizes of 16-32 tokens (in contrast to the 1-2 tokens of prior work with comparable speedup).\r\n\r\nAdditionally, it includes Sparse-Marlin, an extension of the MARLIN kernels adding support to 2:4 weight sparsity, achieving 5.3x speedups on NVIDIA GPUs (Ampere/Ada).","lang":"eng"}],"type":"research_data_reference","has_accepted_license":"1","year":"2024","_id":"19884","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","article_processing_charge":"No","publisher":"Zenodo","day":"24","title":"MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models","citation":{"short":"E. Frantar, R. Castro, J. Chen, T. Hoefler, D.-A. Alistarh, (2024).","apa":"Frantar, E., Castro, R., Chen, J., Hoefler, T., &#38; Alistarh, D.-A. (2024). MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models. Zenodo. <a href=\"https://doi.org/10.5281/ZENODO.14213091\">https://doi.org/10.5281/ZENODO.14213091</a>","ama":"Frantar E, Castro R, Chen J, Hoefler T, Alistarh D-A. MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models. 2024. doi:<a href=\"https://doi.org/10.5281/ZENODO.14213091\">10.5281/ZENODO.14213091</a>","mla":"Frantar, Elias, et al. <i>MARLIN: Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models</i>. Zenodo, 2024, doi:<a href=\"https://doi.org/10.5281/ZENODO.14213091\">10.5281/ZENODO.14213091</a>.","ista":"Frantar E, Castro R, Chen J, Hoefler T, Alistarh D-A. 2024. MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models, Zenodo, <a href=\"https://doi.org/10.5281/ZENODO.14213091\">10.5281/ZENODO.14213091</a>.","ieee":"E. Frantar, R. Castro, J. Chen, T. Hoefler, and D.-A. Alistarh, “MARLIN: Mixed-precision auto-regressive parallel inference on Large Language Models.” Zenodo, 2024.","chicago":"Frantar, Elias, Roberto Castro, Jiale Chen, Torsten Hoefler, and Dan-Adrian Alistarh. “MARLIN: Mixed-Precision Auto-Regressive Parallel Inference on Large Language Models.” Zenodo, 2024. <a href=\"https://doi.org/10.5281/ZENODO.14213091\">https://doi.org/10.5281/ZENODO.14213091</a>."},"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"date_created":"2025-06-24T06:09:18Z","date_updated":"2025-09-30T13:41:56Z","month":"11","oa":1,"date_published":"2024-11-24T00:00:00Z","OA_place":"repository","author":[{"first_name":"Elias","last_name":"Frantar","full_name":"Frantar, Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"last_name":"Castro","full_name":"Castro, Roberto","first_name":"Roberto"},{"id":"4d0a9064-1ff6-11ee-9fa6-ec046c604785","orcid":"0000-0001-5337-5875","first_name":"Jiale","last_name":"Chen","full_name":"Chen, Jiale"},{"first_name":"Torsten","full_name":"Hoefler, Torsten","last_name":"Hoefler"},{"orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","first_name":"Dan-Adrian","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian"}]},{"language":[{"iso":"eng"}],"day":"01","citation":{"short":"I. Markov, K. Alimohammadi, E. Frantar, D.-A. Alistarh, in:, P. Gibbons, G. Pekhimenko, C. De Sa (Eds.), Proceedings of Machine Learning and Systems , Association for Computing Machinery, 2024.","apa":"Markov, I., Alimohammadi, K., Frantar, E., &#38; Alistarh, D.-A. (2024). L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning. In P. Gibbons, G. Pekhimenko, &#38; C. De Sa (Eds.), <i>Proceedings of Machine Learning and Systems </i> (Vol. 6). Athens, Greece: Association for Computing Machinery.","ama":"Markov I, Alimohammadi K, Frantar E, Alistarh D-A. L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning. In: Gibbons P, Pekhimenko G, De Sa C, eds. <i>Proceedings of Machine Learning and Systems </i>. Vol 6. Association for Computing Machinery; 2024.","mla":"Markov, Ilia, et al. “L-GreCo: Layerwise-Adaptive Gradient Compression for Efficient Data-Parallel Deep Learning.” <i>Proceedings of Machine Learning and Systems </i>, edited by P. Gibbons et al., vol. 6, Association for Computing Machinery, 2024.","ista":"Markov I, Alimohammadi K, Frantar E, Alistarh D-A. 2024. L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning. Proceedings of Machine Learning and Systems . MLSys: Machine Learning and Systems vol. 6.","ieee":"I. Markov, K. Alimohammadi, E. Frantar, and D.-A. Alistarh, “L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning,” in <i>Proceedings of Machine Learning and Systems </i>, Athens, Greece, 2024, vol. 6.","chicago":"Markov, Ilia, Kaveh Alimohammadi, Elias Frantar, and Dan-Adrian Alistarh. “L-GreCo: Layerwise-Adaptive Gradient Compression for Efficient Data-Parallel Deep Learning.” In <i>Proceedings of Machine Learning and Systems </i>, edited by P. Gibbons, G. Pekhimenko, and C. De Sa, Vol. 6. Association for Computing Machinery, 2024."},"date_updated":"2026-06-18T17:55:24Z","month":"04","date_published":"2024-04-01T00:00:00Z","main_file_link":[{"open_access":"1","url":"https://proceedings.mlsys.org/paper_files/paper/2024/hash/9069a8976ff06f6443e7f4172990a580-Abstract-Conference.html"}],"abstract":[{"lang":"eng","text":"Data-parallel distributed training of deep neural networks (DNN) has gained very widespread adoption, but can still experience communication bottlenecks. To address this issue, entire families of compression mechanisms have been developed, including quantization, sparsification, and low-rank approximation, some of which are seeing significant practical adoption. Despite this progress, almost all known compression schemes apply compression uniformly across DNN layers, although layers are heterogeneous in terms of parameter count and their impact on model accuracy.In this work, we provide a general framework for adapting the degree of compression across the model's layers dynamically during training, improving the overall compression, while leading to substantial speedups, without sacrificing accuracy. Our framework, called L-GreCo, is based on an adaptive algorithm, which automatically picks the optimal compression parameters for model layers guaranteeing the best compression ratio while satisfying an error constraint. Extensive experiments over image classification and language modeling tasks shows that L-GreCo is effective across all existing families of compression methods, and achieves up to 2.5\r\n×\r\n training speedup and up to 5\r\n×\r\n compression improvement over efficient implementations of existing approaches, while recovering full accuracy. Moreover, L-GreCo is complementary to existing adaptive algorithms, improving their compression ratio by 50\\% and practical throughput by 66\\%. An anonymized implementation is available at https://github.com/LGrCo/L-GreCo."}],"year":"2024","publication_status":"published","article_processing_charge":"No","conference":{"name":"MLSys: Machine Learning and Systems","location":"Athens, Greece","end_date":"2024-04-22","start_date":"2024-04-22"},"publisher":"Association for Computing Machinery","title":"L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning","external_id":{"arxiv":["2210.17357"]},"arxiv":1,"date_created":"2024-08-22T08:29:25Z","oa":1,"author":[{"id":"D0CF4148-C985-11E9-8066-0BDEE5697425","last_name":"Markov","full_name":"Markov, Ilia","first_name":"Ilia"},{"last_name":"Alimohammadi","full_name":"Alimohammadi, Kaveh","first_name":"Kaveh"},{"first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian"}],"quality_controlled":"1","editor":[{"full_name":"Gibbons, P.","last_name":"Gibbons","first_name":"P."},{"first_name":"G.","last_name":"Pekhimenko","full_name":"Pekhimenko, G."},{"first_name":"C.","last_name":"De Sa","full_name":"De Sa, C."}],"oa_version":"Published Version","ddc":["000"],"related_material":{"record":[{"relation":"dissertation_contains","status":"public","id":"17490"}]},"status":"public","corr_author":"1","intvolume":"         6","department":[{"_id":"DaAl"}],"type":"conference","volume":6,"publication":"Proceedings of Machine Learning and Systems ","_id":"17456","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87"},{"status":"public","corr_author":"1","oa_version":"Preprint","scopus_import":"1","department":[{"_id":"DaAl"},{"_id":"GradSch"}],"intvolume":"       235","type":"conference","page":"12284-12303","volume":235,"publication_identifier":{"eissn":["2640-3498"]},"publication":"Proceedings of the 41st International Conference on Machine Learning","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"18113","publisher":"ML Research Press","conference":{"name":"ICML: International Conference on Machine Learning","location":"Vienna, Austria","end_date":"2024-07-27","start_date":"2024-07-21"},"title":"Extreme compression of large language models via additive quantization","arxiv":1,"external_id":{"arxiv":["2401.06118"]},"date_created":"2024-09-22T22:01:43Z","oa":1,"author":[{"first_name":"Vage","last_name":"Egiazarian","full_name":"Egiazarian, Vage"},{"id":"2c18daae-4dbe-11ef-8491-98ce2d960f09","first_name":"Andrei","full_name":"Panferov, Andrei","last_name":"Panferov"},{"full_name":"Kuznedelev, Denis","last_name":"Kuznedelev","first_name":"Denis"},{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","first_name":"Elias","last_name":"Frantar","full_name":"Frantar, Elias"},{"first_name":"Artem","last_name":"Babenko","full_name":"Babenko, Artem"},{"id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","first_name":"Dan-Adrian"}],"acknowledgement":"Authors would like to thank Ruslan Svirschevski for his help in solving technical issues with AQLM and baselines. We also thank Tim Dettmers for helpful discussions on the structure of weights in modern LLMs and size-accuracy trade-offs. The authors would also like to thank Daniil Pavlov for his assistance with CPU benchmarking. Finally, authors would like to thank the communities of ML enthusiasts known as LocalLLaMA5 and Petals community on discord6\r\nfor the crowd wisdom about running LLMs on consumer devices. Egiazarian Vage and Denis Kuznedelev and Andrei Panferov were supported by the grant for research centers in the field of AI provided by the Analytical Center for the Government of the Russian Federation (ACRF) in\r\naccordance with the agreement on the provision of subsidies (identifier of the agreement 000000D730321P5Q0002) and the agreement with HSE University No. 70-2021-00139.","quality_controlled":"1","main_file_link":[{"open_access":"1","url":" https://doi.org/10.48550/arXiv.2401.06118"}],"year":"2024","alternative_title":["PMLR"],"abstract":[{"text":"The emergence of accurate open large language models (LLMs) has led to a race towards performant quantization techniques which can enable their execution on end-user devices. In this paper, we revisit the problem of “extreme” LLM compression—defined as targeting extremely low bit counts, such as 2 to 3 bits per parameter—from the point of view of classic methods in Multi-Codebook Quantization (MCQ). Our algorithm, called AQLM, generalizes the classic Additive Quantization (AQ) approach for information retrieval to advance the state-of-the-art in LLM compression, via two innovations: 1) learned additive quantization of weight matrices in input-adaptive fashion, and 2) joint optimization of codebook parameters across each transformer blocks. Broadly, AQLM is the first scheme that is Pareto optimal in terms of accuracy-vs-model-size when compressing to less than 3 bits per parameter, and significantly improves upon all known schemes in the extreme compression (2bit) regime. In addition, AQLM is practical: we provide fast GPU and CPU implementations of AQLM for token generation, which enable us to match or outperform optimized FP16 implementations for speed, while executing in a much smaller memory footprint.","lang":"eng"}],"publication_status":"published","article_processing_charge":"No","language":[{"iso":"eng"}],"citation":{"apa":"Egiazarian, V., Panferov, A., Kuznedelev, D., Frantar, E., Babenko, A., &#38; Alistarh, D.-A. (2024). Extreme compression of large language models via additive quantization. In <i>Proceedings of the 41st International Conference on Machine Learning</i> (Vol. 235, pp. 12284–12303). Vienna, Austria: ML Research Press.","short":"V. Egiazarian, A. Panferov, D. Kuznedelev, E. Frantar, A. Babenko, D.-A. Alistarh, in:, Proceedings of the 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 12284–12303.","chicago":"Egiazarian, Vage, Andrei Panferov, Denis Kuznedelev, Elias Frantar, Artem Babenko, and Dan-Adrian Alistarh. “Extreme Compression of Large Language Models via Additive Quantization.” In <i>Proceedings of the 41st International Conference on Machine Learning</i>, 235:12284–303. ML Research Press, 2024.","ieee":"V. Egiazarian, A. Panferov, D. Kuznedelev, E. Frantar, A. Babenko, and D.-A. Alistarh, “Extreme compression of large language models via additive quantization,” in <i>Proceedings of the 41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 12284–12303.","ista":"Egiazarian V, Panferov A, Kuznedelev D, Frantar E, Babenko A, Alistarh D-A. 2024. Extreme compression of large language models via additive quantization. Proceedings of the 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 12284–12303.","mla":"Egiazarian, Vage, et al. “Extreme Compression of Large Language Models via Additive Quantization.” <i>Proceedings of the 41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 12284–303.","ama":"Egiazarian V, Panferov A, Kuznedelev D, Frantar E, Babenko A, Alistarh D-A. Extreme compression of large language models via additive quantization. In: <i>Proceedings of the 41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:12284-12303."},"day":"01","date_updated":"2024-10-01T08:13:05Z","date_published":"2024-09-01T00:00:00Z","month":"09"},{"citation":{"mla":"Moakhar, Arshia Soltani, et al. “SPADE: Sparsity-Guided Debugging for Deep Neural Networks.” <i>Proceedings of the 41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 45955–87.","ama":"Moakhar AS, Iofinova EB, Frantar E, Alistarh D-A. SPADE: Sparsity-guided debugging for deep neural networks. In: <i>Proceedings of the 41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:45955-45987.","ieee":"A. S. Moakhar, E. B. Iofinova, E. Frantar, and D.-A. Alistarh, “SPADE: Sparsity-guided debugging for deep neural networks,” in <i>Proceedings of the 41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 45955–45987.","chicago":"Moakhar, Arshia Soltani, Eugenia B Iofinova, Elias Frantar, and Dan-Adrian Alistarh. “SPADE: Sparsity-Guided Debugging for Deep Neural Networks.” In <i>Proceedings of the 41st International Conference on Machine Learning</i>, 235:45955–87. ML Research Press, 2024.","ista":"Moakhar AS, Iofinova EB, Frantar E, Alistarh D-A. 2024. SPADE: Sparsity-guided debugging for deep neural networks. Proceedings of the 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 45955–45987.","short":"A.S. Moakhar, E.B. Iofinova, E. Frantar, D.-A. Alistarh, in:, Proceedings of the 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 45955–45987.","apa":"Moakhar, A. S., Iofinova, E. B., Frantar, E., &#38; Alistarh, D.-A. (2024). SPADE: Sparsity-guided debugging for deep neural networks. In <i>Proceedings of the 41st International Conference on Machine Learning</i> (Vol. 235, pp. 45955–45987). Vienna, Austria: ML Research Press."},"day":"01","language":[{"iso":"eng"}],"date_published":"2024-09-01T00:00:00Z","month":"09","date_updated":"2026-07-27T12:50:03Z","main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2310.04519"}],"article_processing_charge":"No","year":"2024","alternative_title":["PMLR"],"abstract":[{"text":"It is known that sparsity can improve interpretability for deep neural networks. However, existing methods in the area either require networks that are pre-trained with sparsity constraints, or impose sparsity after the fact, altering the network’s general behavior. In this paper, we demonstrate, for the first time, that sparsity can instead be incorporated into the interpretation process itself, as a sample-specific preprocessing step. Unlike previous work, this approach, which we call SPADE, does not place constraints on the trained model and does not affect its behavior during inference on the sample. Given a trained model and a target sample, SPADE uses sample-targeted pruning to provide a \"trace\" of the network’s execution on the sample, reducing the network to the most important connections prior to computing an interpretation. We demonstrate that preprocessing with SPADE significantly increases the accuracy of image saliency maps across several interpretability methods. Additionally, SPADE improves the usefulness of neuron visualizations, aiding humans in reasoning about network behavior. Our code is available at https://github.com/IST-DASLab/SPADE.","lang":"eng"}],"publication_status":"published","title":"SPADE: Sparsity-guided debugging for deep neural networks","external_id":{"arxiv":["2310.04519"]},"arxiv":1,"publisher":"ML Research Press","conference":{"name":"ICML: International Conference on Machine Learning","end_date":"2024-07-27","start_date":"2024-07-21","location":"Vienna, Austria"},"oa":1,"quality_controlled":"1","acknowledgement":"The authors would like to thank Stephen Casper and Tony Wang for their feedback on this work, and Eldar Kurtic for his advice on aspects of the project. This research was supported by the Scientific Service Units (SSU) of IST Austria through resources provided by Scientific Computing (SciComp). EI was supported in part by the FWF DK VGSCO, grant agreement number W1260-N35.","author":[{"first_name":"Arshia Soltani","last_name":"Moakhar","full_name":"Moakhar, Arshia Soltani"},{"full_name":"Iofinova, Eugenia B","last_name":"Iofinova","first_name":"Eugenia B","orcid":"0000-0002-7778-3221","id":"f9a17499-f6e0-11ea-865d-fdf9a3f77117"},{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar"},{"orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","first_name":"Dan-Adrian","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh"}],"date_created":"2024-09-22T22:01:46Z","project":[{"grant_number":"W1260-N35","name":"Vienna Graduate School on Computational Optimization","_id":"9B9290DE-BA93-11EA-9121-9846C619BF3A"}],"department":[{"_id":"DaAl"}],"intvolume":"       235","status":"public","corr_author":"1","related_material":{"link":[{"relation":"software","url":"https://github.com/IST-DASLab/SPADE"}],"record":[{"id":"21854","status":"public","relation":"dissertation_contains"}]},"oa_version":"Preprint","scopus_import":"1","publication":"Proceedings of the 41st International Conference on Machine Learning","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"18121","acknowledged_ssus":[{"_id":"ScienComp"}],"type":"conference","publication_identifier":{"eissn":["2640-3498"]},"page":"45955-45987","volume":235},{"language":[{"iso":"eng"}],"day":"05","citation":{"ieee":"E. Frantar, “Compressing large neural networks: Algorithms, systems and scaling laws,” Institute of Science and Technology Austria, 2024.","chicago":"Frantar, Elias. “Compressing Large Neural Networks: Algorithms, Systems and Scaling Laws.” Institute of Science and Technology Austria, 2024. <a href=\"https://doi.org/10.15479/at:ista:17485\">https://doi.org/10.15479/at:ista:17485</a>.","ista":"Frantar E. 2024. Compressing large neural networks: Algorithms, systems and scaling laws. Institute of Science and Technology Austria.","mla":"Frantar, Elias. <i>Compressing Large Neural Networks: Algorithms, Systems and Scaling Laws</i>. Institute of Science and Technology Austria, 2024, doi:<a href=\"https://doi.org/10.15479/at:ista:17485\">10.15479/at:ista:17485</a>.","ama":"Frantar E. Compressing large neural networks: Algorithms, systems and scaling laws. 2024. doi:<a href=\"https://doi.org/10.15479/at:ista:17485\">10.15479/at:ista:17485</a>","apa":"Frantar, E. (2024). <i>Compressing large neural networks: Algorithms, systems and scaling laws</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/at:ista:17485\">https://doi.org/10.15479/at:ista:17485</a>","short":"E. Frantar, Compressing Large Neural Networks: Algorithms, Systems and Scaling Laws, Institute of Science and Technology Austria, 2024."},"date_updated":"2026-07-29T13:48:40Z","month":"09","date_published":"2024-09-05T00:00:00Z","ec_funded":1,"file":[{"file_name":"thesis-final.zip","access_level":"closed","content_type":"application/zip","file_id":"17570","relation":"source_file","date_updated":"2024-09-05T12:04:11Z","creator":"efrantar","file_size":1615167,"date_created":"2024-09-05T12:04:11Z","checksum":"5d785645805a78c5b4ce7cc3df557b09"},{"file_id":"17880","relation":"main_file","success":1,"content_type":"application/pdf","checksum":"a9dd1c2d23734986924eb44ebb55fd8f","date_created":"2024-09-06T16:24:59Z","date_updated":"2024-09-06T16:24:59Z","creator":"efrantar","file_size":2376611,"access_level":"open_access","file_name":"frantar_thesis_final.pdf"}],"doi":"10.15479/at:ista:17485","supervisor":[{"full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian","orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87"}],"abstract":[{"lang":"eng","text":"Large language models (LLMs) have made tremendous progress in the past few years, from being able to generate coherent text to matching or surpassing humans in a wide variety of creative, knowledge or reasoning tasks. Much of this can be attributed to massively increased scale, both in the size of the model as well as the amount of training data, from 100s of millions to 100s of billions, or even trillions. This trend is expected to continue, which, although exciting, also raises major practical concerns. Already today's 100+ billion parameter LLMs require top-of-the-line hardware just to run. Hence, it is clear that sustaining these developments will require significant efficiency advances.\r\n\r\nHistorically, one of the most practical ways of improving model efficiency has been compression, especially in the form of sparsity or quantization. While this has been studied extensively in the past, existing accurate methods are all designed for models around 100 million parameters; scaling them up to ones literally 1000x larger is highly challenging. In this thesis, we introduce a new unified sparsification and quantization approach OBC, which through additional algorithmic enhancements leads to GPTQ and SparseGPT, the first techniques fast and accurate enough to compress 100+ billion parameter models to 4- or even 3-bit precision and 50% weight-sparsity, respectively. Additionally, we show how weight-only quantizion does not just bring space savings but also up to 4.5x faster generation speed, via custom GPU kernels.\r\n\r\nIn fact, we show for the first time that it is possible to develop an FP16 times INT4 mixed-precision matrix multiplication kernel, called Marlin, which comes close to simultaneously maximizing both memory and compute utilization, making weight-only quantization highly practical even for multi-user serving. Further, we demonstrate that GPTQ can be scaled to widely overparametrized trillion-parameter models, where extreme sub-1-bit compression rates can be achieved without any inference slow-down, by co-designing a bespoke entropy coding scheme together with an efficient kernel.\r\n\r\nFinally, we also study compression from the perspective of someone with access to massive amounts of compute resources for training large models completely from scratch. Here the key questions evolve around the joint scaling behavior between compression, model size, and amount of training data used. Based on extensive experimental results for both vision and text models, we introduce the first scaling law which accurately captures the relationship between weight-sparsity, number of non-zero weights and data. This further allows us to characterize the optimal sparsity, which we find to increase the longer a fixed cost model is being trained.\r\n\r\nOverall, this thesis presents contributions to three different angles of large model efficiency: affordable but accurate algorithms, highly efficient systems implementations, and fundamental scaling laws for compressed training."}],"year":"2024","alternative_title":["ISTA Thesis"],"publication_status":"published","article_processing_charge":"No","publisher":"Institute of Science and Technology Austria","title":"Compressing large neural networks: Algorithms, systems and scaling laws","date_created":"2024-09-02T11:01:48Z","project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","name":"Elastic Coordination for Scalable Machine Learning","grant_number":"805223","call_identifier":"H2020"}],"oa":1,"OA_place":"publisher","author":[{"first_name":"Elias","last_name":"Frantar","full_name":"Frantar, Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"}],"oa_version":"Published Version","ddc":["000"],"status":"public","related_material":{"record":[{"relation":"part_of_dissertation","status":"public","id":"17378"},{"id":"17087","status":"public","relation":"part_of_dissertation"},{"id":"14458","status":"public","relation":"part_of_dissertation"},{"status":"public","id":"18062","relation":"part_of_dissertation"},{"id":"18061","status":"public","relation":"part_of_dissertation"}]},"corr_author":"1","doi_confirm":"1","file_date_updated":"2024-09-06T16:24:59Z","department":[{"_id":"GradSch"},{"_id":"DaAl"}],"degree_awarded":"PhD","type":"dissertation","has_accepted_license":"1","acknowledged_ssus":[{"_id":"ScienComp"}],"publication_identifier":{"issn":["2663-337X"]},"page":"129","_id":"17485","user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9"},{"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","article_processing_charge":"No","_id":"18062","publication":"The Twelfth International Conference on Learning Representations","publication_status":"published","year":"2024","abstract":[{"lang":"eng","text":"We explore the impact of parameter sparsity on the scaling behavior of Transformers trained on massive datasets (i.e., \"foundation models\"), in both vision and language domains. In this setting, we identify the first scaling law describing the relationship between weight sparsity, number of non-zero parameters, and amount of training data, which we validate empirically across model and data scales; on ViT/JFT-4B and T5/C4. These results allow us to characterize the \"optimal sparsity\", the sparsity level which yields the best performance for a given effective model size and training budget. For a fixed number of non-zero parameters, we identify that the optimal sparsity increases with the amount of data used for training. We also extend our study to different sparsity structures (such as the hardware-friendly n:m pattern) and strategies (such as starting from a pretrained dense model). Our findings shed light on the power and limitations of weight sparsity across various parameter and computational settings, offering both theoretical understanding and practical implications for leveraging sparsity towards computational efficiency improvements. We provide pruning and scaling law fitting code at: github.com/google-research/jaxpruner/tree/main/jaxpruner/projects/bigsparse."}],"type":"conference","department":[{"_id":"DaAl"}],"scopus_import":"1","corr_author":"1","ddc":["000"],"related_material":{"record":[{"status":"public","id":"17485","relation":"dissertation_contains"}]},"status":"public","main_file_link":[{"url":"https://openreview.net/forum?id=i9K2ZWkYIP","open_access":"1"}],"oa_version":"Published Version","author":[{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar"},{"full_name":"Ruiz, Carlos Riquelme","last_name":"Ruiz","first_name":"Carlos Riquelme"},{"last_name":"Houlsby","full_name":"Houlsby, Neil","first_name":"Neil"},{"orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian"},{"first_name":"Utku","full_name":"Evci, Utku","last_name":"Evci"}],"quality_controlled":"1","date_published":"2024-01-16T00:00:00Z","oa":1,"month":"01","date_updated":"2026-07-29T13:48:40Z","date_created":"2024-09-13T10:31:08Z","external_id":{"arxiv":["2309.08520"]},"arxiv":1,"citation":{"apa":"Frantar, E., Ruiz, C. R., Houlsby, N., Alistarh, D.-A., &#38; Evci, U. (2024). Scaling laws for sparsely-connected foundation models. In <i>The Twelfth International Conference on Learning Representations</i>. Vienna, Austria.","short":"E. Frantar, C.R. Ruiz, N. Houlsby, D.-A. Alistarh, U. Evci, in:, The Twelfth International Conference on Learning Representations, 2024.","ista":"Frantar E, Ruiz CR, Houlsby N, Alistarh D-A, Evci U. 2024. Scaling laws for sparsely-connected foundation models. The Twelfth International Conference on Learning Representations. ICLR: International Conference on Learning Representations.","ieee":"E. Frantar, C. R. Ruiz, N. Houlsby, D.-A. Alistarh, and U. Evci, “Scaling laws for sparsely-connected foundation models,” in <i>The Twelfth International Conference on Learning Representations</i>, Vienna, Austria, 2024.","chicago":"Frantar, Elias, Carlos Riquelme Ruiz, Neil Houlsby, Dan-Adrian Alistarh, and Utku Evci. “Scaling Laws for Sparsely-Connected Foundation Models.” In <i>The Twelfth International Conference on Learning Representations</i>, 2024.","ama":"Frantar E, Ruiz CR, Houlsby N, Alistarh D-A, Evci U. Scaling laws for sparsely-connected foundation models. In: <i>The Twelfth International Conference on Learning Representations</i>. ; 2024.","mla":"Frantar, Elias, et al. “Scaling Laws for Sparsely-Connected Foundation Models.” <i>The Twelfth International Conference on Learning Representations</i>, 2024."},"day":"16","title":"Scaling laws for sparsely-connected foundation models","conference":{"location":"Vienna, Austria","end_date":"2024-05-07","start_date":"2024-05-07","name":"ICLR: International Conference on Learning Representations"},"language":[{"iso":"eng"}]},{"citation":{"short":"E. Frantar, D.-A. Alistarh, in:, Proceedings of Machine Learning and Systems, 2024.","apa":"Frantar, E., &#38; Alistarh, D.-A. (2024). QMoE: Sub-1-bit compression of trillion parameter models. In <i>Proceedings of Machine Learning and Systems</i> (Vol. 6). Santa Clara, CA, United States.","mla":"Frantar, Elias, and Dan-Adrian Alistarh. “QMoE: Sub-1-Bit Compression of Trillion Parameter Models.” <i>Proceedings of Machine Learning and Systems</i>, vol. 6, 2024.","ama":"Frantar E, Alistarh D-A. QMoE: Sub-1-bit compression of trillion parameter models. In: <i>Proceedings of Machine Learning and Systems</i>. Vol 6. ; 2024.","chicago":"Frantar, Elias, and Dan-Adrian Alistarh. “QMoE: Sub-1-Bit Compression of Trillion Parameter Models.” In <i>Proceedings of Machine Learning and Systems</i>, Vol. 6, 2024.","ieee":"E. Frantar and D.-A. Alistarh, “QMoE: Sub-1-bit compression of trillion parameter models,” in <i>Proceedings of Machine Learning and Systems</i>, Santa Clara, CA, United States, 2024, vol. 6.","ista":"Frantar E, Alistarh D-A. 2024. QMoE: Sub-1-bit compression of trillion parameter models. Proceedings of Machine Learning and Systems. MLSys: Machine Learning and Systems vol. 6."},"day":"01","title":"QMoE: Sub-1-bit compression of trillion parameter models","conference":{"start_date":"2024-05-13","end_date":"2024-05-16","location":"Santa Clara, CA, United States","name":"MLSys: Machine Learning and Systems"},"language":[{"iso":"eng"}],"author":[{"first_name":"Elias","full_name":"Frantar, Elias","last_name":"Frantar","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"first_name":"Dan-Adrian","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X"}],"quality_controlled":"1","das_tickbox":"1","date_published":"2024-05-01T00:00:00Z","oa":1,"month":"05","date_updated":"2026-07-29T13:48:40Z","date_created":"2024-09-13T10:01:38Z","department":[{"_id":"DaAl"}],"intvolume":"         6","ddc":["000"],"related_material":{"record":[{"relation":"dissertation_contains","status":"public","id":"17485"}]},"status":"public","corr_author":"1","main_file_link":[{"url":"https://proceedings.mlsys.org/paper_files/paper/2024/hash/c74b624843218d9b6713fcf299d6d5e4-Abstract-Conference.html","open_access":"1"}],"oa_version":"Published Version","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","article_processing_charge":"No","_id":"18061","publication":"Proceedings of Machine Learning and Systems","publication_status":"published","volume":6,"year":"2024","type":"conference","abstract":[{"lang":"eng","text":"Mixture-of-Experts (MoE) architectures offer a general solution to the high inference costs of large language models (LLMs) via sparse routing, bringing faster and more accurate models, at the cost of massive parameter counts. For example, the SwitchTransformer-c2048 model has 1.6 trillion parameters, requiring 3.2TB of accelerator memory to run efficiently, which makes practical deployment challenging and expensive. In this paper, we present a solution to this memory problem, in form of a new compression and execution framework called QMoE. Specifically, QMoE consists of a scalable algorithm which accurately compresses trillion-parameter MoEs to less than 1 bit per parameter, in a custom format co-designed with bespoke GPU decoding kernels to facilitate efficient end-to-end compressed inference, with minor runtime overheads relative to uncompressed execution. Concretely, QMoE can compress the 1.6 trillion parameter SwitchTransformer-c2048 model to less than 160GB (20x compression, 0.8 bits per parameter) at only minor accuracy loss, in less than a day on a single GPU. This enables, for the first time, the execution of a trillion-parameter model on affordable commodity hardware, like a single server with 4x NVIDIA A6000 or 8x NVIDIA 3090 GPUs, at less than 5% runtime overhead relative to ideal uncompressed inference. The anonymized code is available at: github.com/mlsys24-qmoe/qmoe."}]},{"publication":"11th International Conference on Learning Representations ","_id":"17378","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","type":"conference","has_accepted_license":"1","acknowledged_ssus":[{"_id":"ScienComp"}],"file_date_updated":"2024-08-05T07:52:44Z","department":[{"_id":"DaAl"}],"oa_version":"Published Version","related_material":{"record":[{"relation":"dissertation_contains","id":"17485","status":"public"}],"link":[{"url":"https://github.com/IST-DASLab/gptq","relation":"software"}]},"corr_author":"1","status":"public","ddc":["000"],"scopus_import":"1","oa":1,"acknowledgement":"Elias Frantar and Dan Alistarh gratefully acknowledge funding from the European Research Council (ERC) under the European Union’s Horizon 2020 programme (grant agreement No. 805223 ScaleML), as well as experimental support from Eldar Kurtic, and from the IST Austria IT department, in particular Stefano Elefante, Andrei Hornoiu, and Alois Schloegl. The work of Saleh Ashkboos and Torsten Hoefler was supported by the PASC DaCeMI project, received EuroHPC-JU funding under grant MAELSTROM, No. 955513. We thank the Swiss National Supercomputing Center (CSCS) for supporting us with compute infrastructure.","author":[{"last_name":"Frantar","full_name":"Frantar, Elias","first_name":"Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"full_name":"Ashkboos, Saleh","last_name":"Ashkboos","first_name":"Saleh"},{"first_name":"Torsten","full_name":"Hoefler, Torsten","last_name":"Hoefler"},{"orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","first_name":"Dan-Adrian","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian"}],"quality_controlled":"1","date_created":"2024-08-04T22:01:22Z","project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","grant_number":"805223","name":"Elastic Coordination for Scalable Machine Learning","call_identifier":"H2020"}],"title":"OPTQ: Accurate post-training quantization for generative pre-trained transformers","conference":{"start_date":"2023-05-01","end_date":"2023-05-05","location":"Kigali, Rwanda","name":"ICLR: International Conference on Learning Representations"},"publisher":"International Conference on Learning Representations","article_processing_charge":"No","abstract":[{"lang":"eng","text":"Generative Pre-trained Transformer models, known as GPT or OPT, set themselves apart through breakthrough performance across complex language modelling tasks, but also by their extremely high computational and storage costs. Specifically, due to their massive size, even inference for large, highly-accurate GPT models may require multiple performant GPUs, which limits the usability of such models. While there is emerging work on relieving this pressure via model compression, the applicability and performance of existing compression techniques is limited by the scale and complexity of GPT models. In this paper, we address this challenge, and propose OPTQ, a new one-shot weight quantization method based on approximate second-order information, that is both highly-accurate and highly-efficient. Specifically, OPTQ can quantize GPT models with 175 billion parameters in approximately four GPU hours, reducing the bitwidth down to 3 or 4 bits per weight, with negligible accuracy degradation relative to the uncompressed baseline. Our method more than doubles the compression gains relative to previously-proposed one-shot quantization methods, preserving accuracy, allowing us for the first time to execute an 175 billion-parameter model inside a single GPU for generative inference. Moreover, we also show that our method can still provide reasonable accuracy in the extreme quantization regime, in which weights are quantized to 2-bit or even ternary quantization levels. We show experimentally that these improvements can be leveraged for end-to-end inference speedups over FP16, of around 3.25x when using high-end GPUs (NVIDIA A100) and 4.5x when using more cost-effective ones (NVIDIA A6000). The implementation is available at https://github.com/IST-DASLab/gptq."}],"year":"2023","publication_status":"published","month":"05","date_published":"2023-05-01T00:00:00Z","ec_funded":1,"file":[{"file_name":"2023_ICLR_Frantar.pdf","relation":"main_file","file_id":"17385","content_type":"application/pdf","success":1,"checksum":"aacbf11dbd8b02a3e0bfd942a33e0593","date_created":"2024-08-05T07:52:44Z","creator":"dernst","file_size":437492,"date_updated":"2024-08-05T07:52:44Z","access_level":"open_access"}],"date_updated":"2026-07-29T13:48:39Z","day":"01","citation":{"ista":"Frantar E, Ashkboos S, Hoefler T, Alistarh D-A. 2023. OPTQ: Accurate post-training quantization for generative pre-trained transformers. 11th International Conference on Learning Representations . ICLR: International Conference on Learning Representations.","ieee":"E. Frantar, S. Ashkboos, T. Hoefler, and D.-A. Alistarh, “OPTQ: Accurate post-training quantization for generative pre-trained transformers,” in <i>11th International Conference on Learning Representations </i>, Kigali, Rwanda, 2023.","chicago":"Frantar, Elias, Saleh Ashkboos, Torsten Hoefler, and Dan-Adrian Alistarh. “OPTQ: Accurate Post-Training Quantization for Generative Pre-Trained Transformers.” In <i>11th International Conference on Learning Representations </i>. International Conference on Learning Representations, 2023.","ama":"Frantar E, Ashkboos S, Hoefler T, Alistarh D-A. OPTQ: Accurate post-training quantization for generative pre-trained transformers. In: <i>11th International Conference on Learning Representations </i>. International Conference on Learning Representations; 2023.","mla":"Frantar, Elias, et al. “OPTQ: Accurate Post-Training Quantization for Generative Pre-Trained Transformers.” <i>11th International Conference on Learning Representations </i>, International Conference on Learning Representations, 2023.","apa":"Frantar, E., Ashkboos, S., Hoefler, T., &#38; Alistarh, D.-A. (2023). OPTQ: Accurate post-training quantization for generative pre-trained transformers. In <i>11th International Conference on Learning Representations </i>. Kigali, Rwanda: International Conference on Learning Representations.","short":"E. Frantar, S. Ashkboos, T. Hoefler, D.-A. Alistarh, in:, 11th International Conference on Learning Representations , International Conference on Learning Representations, 2023."},"language":[{"iso":"eng"}]},{"article_processing_charge":"No","alternative_title":["PMLR"],"year":"2023","abstract":[{"lang":"eng","text":"We show for the first time that large-scale generative pretrained transformer (GPT) family models can be pruned to at least 50% sparsity in one-shot, without any retraining, at minimal loss of accuracy. This is achieved via a new pruning method called SparseGPT, specifically designed to work efficiently and accurately on massive GPT-family models. We can execute SparseGPT on the largest available open-source models, OPT-175B and BLOOM-176B, in under 4.5 hours, and can reach 60% unstructured sparsity with negligible increase in perplexity: remarkably, more than 100 billion weights from these models can be ignored at inference time. SparseGPT generalizes to semi-structured (2:4 and 4:8) patterns, and is compatible with weight quantization approaches. The code is available at: https://github.com/IST-DASLab/sparsegpt."}],"publication_status":"published","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2301.00774","open_access":"1"}],"date_published":"2023-07-30T00:00:00Z","month":"07","ec_funded":1,"date_updated":"2026-07-29T13:48:39Z","citation":{"short":"E. Frantar, D.-A. Alistarh, in:, Proceedings of the 40th International Conference on Machine Learning, ML Research Press, 2023, pp. 10323–10337.","apa":"Frantar, E., &#38; Alistarh, D.-A. (2023). SparseGPT: Massive language models can be accurately pruned in one-shot. In <i>Proceedings of the 40th International Conference on Machine Learning</i> (Vol. 202, pp. 10323–10337). Honolulu, Hawaii, HI, United States: ML Research Press.","ama":"Frantar E, Alistarh D-A. SparseGPT: Massive language models can be accurately pruned in one-shot. In: <i>Proceedings of the 40th International Conference on Machine Learning</i>. Vol 202. ML Research Press; 2023:10323-10337.","mla":"Frantar, Elias, and Dan-Adrian Alistarh. “SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot.” <i>Proceedings of the 40th International Conference on Machine Learning</i>, vol. 202, ML Research Press, 2023, pp. 10323–37.","ista":"Frantar E, Alistarh D-A. 2023. SparseGPT: Massive language models can be accurately pruned in one-shot. Proceedings of the 40th International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 202, 10323–10337.","ieee":"E. Frantar and D.-A. Alistarh, “SparseGPT: Massive language models can be accurately pruned in one-shot,” in <i>Proceedings of the 40th International Conference on Machine Learning</i>, Honolulu, Hawaii, HI, United States, 2023, vol. 202, pp. 10323–10337.","chicago":"Frantar, Elias, and Dan-Adrian Alistarh. “SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot.” In <i>Proceedings of the 40th International Conference on Machine Learning</i>, 202:10323–37. ML Research Press, 2023."},"day":"30","language":[{"iso":"eng"}],"publication":"Proceedings of the 40th International Conference on Machine Learning","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"14458","acknowledged_ssus":[{"_id":"ScienComp"}],"type":"conference","volume":202,"page":"10323-10337","publication_identifier":{"eissn":["2640-3498"]},"department":[{"_id":"DaAl"}],"intvolume":"       202","related_material":{"record":[{"status":"public","id":"17485","relation":"dissertation_contains"}]},"status":"public","corr_author":"1","oa_version":"Preprint","scopus_import":"1","oa":1,"quality_controlled":"1","author":[{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","first_name":"Elias","last_name":"Frantar","full_name":"Frantar, Elias"},{"full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian","orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87"}],"acknowledgement":"The authors gratefully acknowledge funding from the European Research Council (ERC) under the European Union’s Horizon 2020 programme (grant agreement No. 805223 ScaleML), as well as experimental support from Eldar Kurtic, and from the IST Austria IT department, in particular Stefano Elefante, Andrei Hornoiu, and Alois Schloegl.","date_created":"2023-10-29T23:01:16Z","project":[{"call_identifier":"H2020","name":"Elastic Coordination for Scalable Machine Learning","grant_number":"805223","_id":"268A44D6-B435-11E9-9278-68D0E5697425"}],"title":"SparseGPT: Massive language models can be accurately pruned in one-shot","external_id":{"arxiv":["2301.00774"]},"arxiv":1,"publisher":"ML Research Press","conference":{"name":"ICML: International Conference on Machine Learning","start_date":"2023-07-23","end_date":"2023-07-29","location":"Honolulu, Hawaii, HI, United States"}},{"author":[{"last_name":"Frantar","full_name":"Frantar, Elias","first_name":"Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian","orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87"}],"quality_controlled":"1","acknowledgement":"We gratefully acknowledge funding from the European Research Council (ERC) under the European Union’s Horizon 2020 programme (grant agreement No 805223 ScaleML),\r\nas well as computational support from AWS EC2. We thank Eldar Kurtic for code and hyper-parameters for BERT pruning, and the Neural Magic Team, notably Michael Goin and\r\nMark Kurtz, for support with their software.","oa":1,"project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","call_identifier":"H2020","grant_number":"805223","name":"Elastic Coordination for Scalable Machine Learning"}],"date_created":"2024-05-28T13:45:20Z","external_id":{"isi":["000922378801029"]},"title":"SPDY: Accurate pruning with speedup guarantees","conference":{"name":"ICML: International Conference on Machine Learning","location":"Baltimore, MD, United States","end_date":"2022-07-23","start_date":"2022-07-17"},"publisher":"ML Research Press","_id":"17059","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","publication":"39th International Conference on Machine Learning","page":"6726-6743","volume":162,"type":"conference","has_accepted_license":"1","intvolume":"       162","department":[{"_id":"DaAl"}],"isi":1,"file_date_updated":"2024-08-19T06:54:41Z","scopus_import":"1","oa_version":"Published Version","status":"public","corr_author":"1","ddc":["000"],"file":[{"file_name":"2022_PMLR_Frantar.pdf","success":1,"content_type":"application/pdf","file_id":"17440","relation":"main_file","date_updated":"2024-08-19T06:54:41Z","creator":"dernst","file_size":615916,"date_created":"2024-08-19T06:54:41Z","checksum":"5179a1e4dfc0fbfab6674907299e414a","access_level":"open_access"}],"ec_funded":1,"month":"07","date_published":"2022-07-20T00:00:00Z","date_updated":"2025-04-14T07:49:14Z","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"day":"20","citation":{"chicago":"Frantar, Elias, and Dan-Adrian Alistarh. “SPDY: Accurate Pruning with Speedup Guarantees.” In <i>39th International Conference on Machine Learning</i>, 162:6726–43. ML Research Press, 2022.","ieee":"E. Frantar and D.-A. Alistarh, “SPDY: Accurate pruning with speedup guarantees,” in <i>39th International Conference on Machine Learning</i>, Baltimore, MD, United States, 2022, vol. 162, pp. 6726–6743.","ista":"Frantar E, Alistarh D-A. 2022. SPDY: Accurate pruning with speedup guarantees. 39th International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 162, 6726–6743.","mla":"Frantar, Elias, and Dan-Adrian Alistarh. “SPDY: Accurate Pruning with Speedup Guarantees.” <i>39th International Conference on Machine Learning</i>, vol. 162, ML Research Press, 2022, pp. 6726–43.","ama":"Frantar E, Alistarh D-A. SPDY: Accurate pruning with speedup guarantees. In: <i>39th International Conference on Machine Learning</i>. Vol 162. ML Research Press; 2022:6726-6743.","apa":"Frantar, E., &#38; Alistarh, D.-A. (2022). SPDY: Accurate pruning with speedup guarantees. In <i>39th International Conference on Machine Learning</i> (Vol. 162, pp. 6726–6743). Baltimore, MD, United States: ML Research Press.","short":"E. Frantar, D.-A. Alistarh, in:, 39th International Conference on Machine Learning, ML Research Press, 2022, pp. 6726–6743."},"language":[{"iso":"eng"}],"article_processing_charge":"Yes","publication_status":"published","abstract":[{"text":"The recent focus on the efficiency of deep neural networks (DNNs) has led to significant work on model compression approaches, of which weight pruning is one of the most popular. At the same time, there is rapidly-growing computational support for efficiently executing the unstructured-sparse models obtained via pruning. Yet, most existing pruning methods minimize just the number of remaining weights, i.e. the size of the model, rather than optimizing for inference time. We address this gap by introducing SPDY, a new compression method which automatically determines layer-wise sparsity targets achieving a desired inference speedup on a given system, while minimizing accuracy loss. SPDY is the composition of two new techniques. The first is an efficient and general dynamic programming algorithm for solving constrained layer-wise compression problems, given a set of layer-wise error scores. The second technique is a local search procedure for automatically determining such scores in an accurate and robust manner. Experiments across popular vision and language models show that SPDY guarantees speedups while recovering higher accuracy relative to existing strategies, both for one-shot and gradual pruning scenarios, and is compatible with most existing pruning approaches. We also extend our approach to the recently-proposed task of pruning with very little data, where we achieve the best known accuracy recovery when pruning to the GPU-supported 2:4 sparsity pattern.","lang":"eng"}],"alternative_title":["PMLR"],"year":"2022"},{"date_updated":"2024-07-31T11:05:32Z","month":"12","date_published":"2022-12-01T00:00:00Z","file":[{"file_name":"2022_EMNLP_Kurtic.pdf","content_type":"application/pdf","success":1,"relation":"main_file","file_id":"17354","creator":"dernst","date_updated":"2024-07-31T11:03:34Z","file_size":522563,"checksum":"c47b9edd8a9f743ac77a593de6d2e84a","date_created":"2024-07-31T11:03:34Z","access_level":"open_access"}],"language":[{"iso":"eng"}],"day":"01","citation":{"short":"E. Kurtic, D. Campos, T. Nguyen, E. Frantar, M. Kurtz, B. Fineran, M. Goin, D.-A. Alistarh, in:, Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, Association for Computational Linguistics, 2022, pp. 4163–4181.","apa":"Kurtic, E., Campos, D., Nguyen, T., Frantar, E., Kurtz, M., Fineran, B., … Alistarh, D.-A. (2022). The optimal BERT surgeon: Scalable and accurate second-order pruning for large language models. In <i>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</i> (pp. 4163–4181). Abu Dhabi, United Arab Emirates: Association for Computational Linguistics. <a href=\"https://doi.org/10.18653/v1/2022.emnlp-main.279\">https://doi.org/10.18653/v1/2022.emnlp-main.279</a>","ama":"Kurtic E, Campos D, Nguyen T, et al. The optimal BERT surgeon: Scalable and accurate second-order pruning for large language models. In: <i>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</i>. Association for Computational Linguistics; 2022:4163-4181. doi:<a href=\"https://doi.org/10.18653/v1/2022.emnlp-main.279\">10.18653/v1/2022.emnlp-main.279</a>","mla":"Kurtic, Eldar, et al. “The Optimal BERT Surgeon: Scalable and Accurate Second-Order Pruning for Large Language Models.” <i>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</i>, Association for Computational Linguistics, 2022, pp. 4163–81, doi:<a href=\"https://doi.org/10.18653/v1/2022.emnlp-main.279\">10.18653/v1/2022.emnlp-main.279</a>.","ista":"Kurtic E, Campos D, Nguyen T, Frantar E, Kurtz M, Fineran B, Goin M, Alistarh D-A. 2022. The optimal BERT surgeon: Scalable and accurate second-order pruning for large language models. Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. EMNLP: Conference on Empirical Methods in Natural Language Processing, 4163–4181.","chicago":"Kurtic, Eldar, Daniel Campos, Tuan Nguyen, Elias Frantar, Mark Kurtz, Benjamin Fineran, Michael Goin, and Dan-Adrian Alistarh. “The Optimal BERT Surgeon: Scalable and Accurate Second-Order Pruning for Large Language Models.” In <i>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</i>, 4163–81. Association for Computational Linguistics, 2022. <a href=\"https://doi.org/10.18653/v1/2022.emnlp-main.279\">https://doi.org/10.18653/v1/2022.emnlp-main.279</a>.","ieee":"E. Kurtic <i>et al.</i>, “The optimal BERT surgeon: Scalable and accurate second-order pruning for large language models,” in <i>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</i>, Abu Dhabi, United Arab Emirates, 2022, pp. 4163–4181."},"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"abstract":[{"lang":"eng","text":"In this paper, we consider the problem of sparsifying BERT models, which are a key building block for natural language processing, in order to reduce their storage and computational cost. We introduce the Optimal BERT Surgeon (oBERT), an efficient and accurate pruning method based on approximate second-order information, which we show to yield state-of-the-art results in both stages of language tasks: pre-training and fine-tuning. Specifically, oBERT extends existing work on second-order pruning by allowing for pruning weight blocks, and is the first such method that is applicable at BERT scale. Second, we investigate compounding compression approaches to obtain highly compressed but accurate models for deployment on edge devices. These models significantly push boundaries of the current state-of-the-art sparse BERT models with respect to all metrics: model size, inference speed and task accuracy. For example, relative to the dense BERT-base, we obtain 10x model size compression with < 1% accuracy drop, 10x CPU-inference speedup with < 2% accuracy drop, and 29x CPU-inference speedup with < 7.5% accuracy drop. Our code, fully integrated with Transformers and SparseML, is available at https://github.com/neuralmagic/sparseml/tree/main/research/optimal_BERT_surgeon_oBERT."}],"year":"2022","publication_status":"published","article_processing_charge":"Yes","doi":"10.18653/v1/2022.emnlp-main.279","date_created":"2024-05-29T06:40:55Z","oa":1,"author":[{"id":"47beb3a5-07b5-11eb-9b87-b108ec578218","first_name":"Eldar","last_name":"Kurtic","full_name":"Kurtic, Eldar"},{"full_name":"Campos, Daniel","last_name":"Campos","first_name":"Daniel"},{"first_name":"Tuan","full_name":"Nguyen, Tuan","last_name":"Nguyen"},{"first_name":"Elias","last_name":"Frantar","full_name":"Frantar, Elias","id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f"},{"first_name":"Mark","last_name":"Kurtz","full_name":"Kurtz, Mark"},{"first_name":"Benjamin","full_name":"Fineran, Benjamin","last_name":"Fineran"},{"last_name":"Goin","full_name":"Goin, Michael","first_name":"Michael"},{"id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X","first_name":"Dan-Adrian","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh"}],"quality_controlled":"1","conference":{"location":"Abu Dhabi, United Arab Emirates","end_date":"2022-12-11","start_date":"2022-12-07","name":"EMNLP: Conference on Empirical Methods in Natural Language Processing"},"publisher":"Association for Computational Linguistics","title":"The optimal BERT surgeon: Scalable and accurate second-order pruning for large language models","arxiv":1,"external_id":{"arxiv":["2203.07259"]},"type":"conference","has_accepted_license":"1","page":"4163-4181","publication":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing","_id":"17088","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","oa_version":"Published Version","ddc":["000"],"related_material":{"link":[{"relation":"software","url":"https://github.com/neuralmagic/sparseml/tree/main/research/optimal_BERT_surgeon_oBERT"}]},"corr_author":"1","status":"public","scopus_import":"1","file_date_updated":"2024-07-31T11:03:34Z","department":[{"_id":"DaAl"}]},{"oa":1,"acknowledgement":"We gratefully acknowledge funding from the European Research Council (ERC) under the European Union’s Horizon 2020 programme (grant agreement No 805223 ScaleML), as well as computational support from AWS EC2. We thank Eldar Kurtic for providing us BERT code and pretrained models, and the Neural Magic Team, notably Michael Goin and Mark Kurtz, for support with their software. ","author":[{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","full_name":"Frantar, Elias","last_name":"Frantar","first_name":"Elias"},{"id":"DD138E24-D89D-11E9-9DC0-DEF6E5697425","full_name":"Singh, Sidak Pal","last_name":"Singh","first_name":"Sidak Pal"},{"id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X","first_name":"Dan-Adrian","full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh"}],"quality_controlled":"1","date_created":"2024-05-29T06:38:26Z","project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","call_identifier":"H2020","grant_number":"805223","name":"Elastic Coordination for Scalable Machine Learning"}],"title":"Optimal brain compression: A framework for accurate post-training quantization and pruning","external_id":{"arxiv":["2208.11580"]},"arxiv":1,"publisher":"ML Research Press","conference":{"location":"New Orleans, LA, United States","end_date":"2022-12-09","start_date":"2022-11-28","name":"NeurIPS: Neural Information Processing Systems"},"publication":"36th Conference on Neural Information Processing Systems","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"17087","has_accepted_license":"1","type":"conference","volume":35,"publication_identifier":{"isbn":["9781713871088"]},"file_date_updated":"2024-08-05T09:25:39Z","department":[{"_id":"DaAl"}],"intvolume":"        35","related_material":{"record":[{"relation":"dissertation_contains","id":"17485","status":"public"}]},"status":"public","corr_author":"1","ddc":["000"],"oa_version":"Submitted Version","scopus_import":"1","date_published":"2022-12-01T00:00:00Z","month":"12","file":[{"file_name":"2022_NeurIPS_Frantar.pdf","access_level":"open_access","checksum":"38e7d75f578e8d2e207c81895e09f211","date_created":"2024-08-05T09:25:39Z","file_size":491843,"date_updated":"2024-08-05T09:25:39Z","creator":"dernst","relation":"main_file","file_id":"17391","success":1,"content_type":"application/pdf"}],"ec_funded":1,"date_updated":"2026-07-29T13:48:39Z","citation":{"apa":"Frantar, E., Singh, S. P., &#38; Alistarh, D.-A. (2022). Optimal brain compression: A framework for accurate post-training quantization and pruning. In <i>36th Conference on Neural Information Processing Systems</i> (Vol. 35). New Orleans, LA, United States: ML Research Press.","short":"E. Frantar, S.P. Singh, D.-A. Alistarh, in:, 36th Conference on Neural Information Processing Systems, ML Research Press, 2022.","ieee":"E. Frantar, S. P. Singh, and D.-A. Alistarh, “Optimal brain compression: A framework for accurate post-training quantization and pruning,” in <i>36th Conference on Neural Information Processing Systems</i>, New Orleans, LA, United States, 2022, vol. 35.","chicago":"Frantar, Elias, Sidak Pal Singh, and Dan-Adrian Alistarh. “Optimal Brain Compression: A Framework for Accurate Post-Training Quantization and Pruning.” In <i>36th Conference on Neural Information Processing Systems</i>, Vol. 35. ML Research Press, 2022.","ista":"Frantar E, Singh SP, Alistarh D-A. 2022. Optimal brain compression: A framework for accurate post-training quantization and pruning. 36th Conference on Neural Information Processing Systems. NeurIPS: Neural Information Processing Systems, NeurIPS, vol. 35.","mla":"Frantar, Elias, et al. “Optimal Brain Compression: A Framework for Accurate Post-Training Quantization and Pruning.” <i>36th Conference on Neural Information Processing Systems</i>, vol. 35, ML Research Press, 2022.","ama":"Frantar E, Singh SP, Alistarh D-A. Optimal brain compression: A framework for accurate post-training quantization and pruning. In: <i>36th Conference on Neural Information Processing Systems</i>. Vol 35. ML Research Press; 2022."},"day":"01","language":[{"iso":"eng"}],"article_processing_charge":"No","alternative_title":["NeurIPS"],"year":"2022","abstract":[{"lang":"eng","text":"We consider the problem of model compression for deep neural networks (DNNs) in the challenging one-shot/post-training setting, in which we are given an accurate trained model, and must compress it without any retraining, based only on a small amount of calibration input data. This problem has become popular in view of the emerging software and hardware support for executing models compressed via pruning and/or quantization with speedup, and well-performing solutions have been proposed independently for both compression approaches.In this paper, we introduce a new compression framework which covers both weight pruning and quantization in a unified setting, is time- and space-efficient, and considerably improves upon the practical performance of existing post-training methods. At the technical level, our approach is based on an exact and efficient realization of the classical Optimal Brain Surgeon (OBS) framework of [LeCun, Denker, and Solla, 1990] extended to also cover weight quantization at the scale of modern DNNs. From the practical perspective, our experimental results show that it can improve significantly upon the compression-accuracy trade-offs of existing post-training methods, and that it can enable the accurate compound application of both pruning and quantization in a post-training setting."}],"publication_status":"published"},{"article_processing_charge":"No","publication_status":"published","alternative_title":["Advances in Neural Information Processing Systems"],"year":"2021","abstract":[{"text":"Efficiently approximating local curvature information of the loss function is a key tool for optimization and compression of deep neural networks. Yet, most existing methods to approximate second-order information have high computational\r\nor storage costs, which limits their practicality. In this work, we investigate matrix-free, linear-time approaches for estimating Inverse-Hessian Vector Products (IHVPs) for the case when the Hessian can be approximated as a sum of rank-one matrices, as in the classic approximation of the Hessian by the empirical Fisher matrix. We propose two new algorithms: the first is tailored towards network compression and can compute the IHVP for dimension d, if the Hessian is given as a sum of m rank-one matrices, using O(dm2) precomputation, O(dm) cost for computing the IHVP, and query cost O(m) for any single element of the inverse Hessian. The second algorithm targets an optimization setting, where we wish to compute the product between the inverse Hessian, estimated over a sliding window of optimization steps, and a given gradient direction, as required for preconditioned SGD. We give an algorithm with cost O(dm + m2) for computing the IHVP and O(dm + m3) for adding or removing any gradient from the sliding window. These\r\ntwo algorithms yield state-of-the-art results for network pruning and optimization with lower computational overhead relative to existing second-order methods. Implementations are available at [9] and [17].","lang":"eng"}],"main_file_link":[{"url":"https://proceedings.neurips.cc/paper/2021/file/7cfd5df443b4eb0d69886a583b33de4c-Paper.pdf","open_access":"1"}],"ec_funded":1,"date_published":"2021-12-06T00:00:00Z","month":"12","date_updated":"2026-06-18T17:18:44Z","citation":{"ista":"Frantar E, Kurtic E, Alistarh D-A. 2021. M-FAC: Efficient matrix-free approximations of second-order information. 35th Conference on Neural Information Processing Systems. NeurIPS: Neural Information Processing Systems, Advances in Neural Information Processing Systems, vol. 34, 14873–14886.","chicago":"Frantar, Elias, Eldar Kurtic, and Dan-Adrian Alistarh. “M-FAC: Efficient Matrix-Free Approximations of Second-Order Information.” In <i>35th Conference on Neural Information Processing Systems</i>, 34:14873–86. Neural Information Processing Systems Foundation, 2021.","ieee":"E. Frantar, E. Kurtic, and D.-A. Alistarh, “M-FAC: Efficient matrix-free approximations of second-order information,” in <i>35th Conference on Neural Information Processing Systems</i>, Virtual, Online, 2021, vol. 34, pp. 14873–14886.","ama":"Frantar E, Kurtic E, Alistarh D-A. M-FAC: Efficient matrix-free approximations of second-order information. In: <i>35th Conference on Neural Information Processing Systems</i>. Vol 34. Neural Information Processing Systems Foundation; 2021:14873-14886.","mla":"Frantar, Elias, et al. “M-FAC: Efficient Matrix-Free Approximations of Second-Order Information.” <i>35th Conference on Neural Information Processing Systems</i>, vol. 34, Neural Information Processing Systems Foundation, 2021, pp. 14873–86.","apa":"Frantar, E., Kurtic, E., &#38; Alistarh, D.-A. (2021). M-FAC: Efficient matrix-free approximations of second-order information. In <i>35th Conference on Neural Information Processing Systems</i> (Vol. 34, pp. 14873–14886). Virtual, Online: Neural Information Processing Systems Foundation.","short":"E. Frantar, E. Kurtic, D.-A. Alistarh, in:, 35th Conference on Neural Information Processing Systems, Neural Information Processing Systems Foundation, 2021, pp. 14873–14886."},"day":"06","language":[{"iso":"eng"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"11463","publication":"35th Conference on Neural Information Processing Systems","page":"14873-14886","volume":34,"publication_identifier":{"isbn":["9781713845393"],"issn":["1049-5258"]},"type":"conference","department":[{"_id":"DaAl"}],"intvolume":"        34","scopus_import":"1","corr_author":"1","status":"public","ddc":["000"],"oa_version":"Published Version","quality_controlled":"1","author":[{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","full_name":"Frantar, Elias","last_name":"Frantar","first_name":"Elias"},{"id":"47beb3a5-07b5-11eb-9b87-b108ec578218","first_name":"Eldar","full_name":"Kurtic, Eldar","last_name":"Kurtic"},{"full_name":"Alistarh, Dan-Adrian","last_name":"Alistarh","first_name":"Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X"}],"acknowledgement":"We gratefully acknowledge funding the European Research Council (ERC) under the European Union’s Horizon 2020 research and innovation programme (grant agreement No 805223 ScaleML), as well as computational support from Amazon Web Services (AWS) EC2.","oa":1,"project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","name":"Elastic Coordination for Scalable Machine Learning","grant_number":"805223","call_identifier":"H2020"}],"date_created":"2022-06-26T22:01:35Z","external_id":{"arxiv":["2010.08222"]},"arxiv":1,"title":"M-FAC: Efficient matrix-free approximations of second-order information","publisher":"Neural Information Processing Systems Foundation","conference":{"end_date":"2021-12-14","start_date":"2021-12-06","location":"Virtual, Online","name":"NeurIPS: Neural Information Processing Systems"}},{"article_processing_charge":"No","publication_status":"published","year":"2020","abstract":[{"text":"We study the problem of learning from multiple untrusted data sources, a scenario of increasing practical relevance given the recent emergence of crowdsourcing and collaborative learning paradigms. Specifically, we analyze the situation in which a learning system obtains datasets from multiple sources, some of which might be biased or even adversarially perturbed. It is\r\nknown that in the single-source case, an adversary with the power to corrupt a fixed fraction of the training data can prevent PAC-learnability, that is, even in the limit of infinitely much training data, no learning system can approach the optimal test error. In this work we show that, surprisingly, the same is not true in the multi-source setting, where the adversary can arbitrarily\r\ncorrupt a fixed fraction of the data sources. Our main results are a generalization bound that provides finite-sample guarantees for this learning setting, as well as corresponding lower bounds. Besides establishing PAC-learnability our results also show that in a cooperative learning setting sharing data with other parties has provable benefits, even if some\r\nparticipants are malicious. ","lang":"eng"}],"file":[{"file_id":"9120","relation":"main_file","success":1,"content_type":"application/pdf","date_created":"2021-02-15T09:00:01Z","checksum":"cc755d0054bc4b2be778ea7aa7884d2f","date_updated":"2021-02-15T09:00:01Z","file_size":281286,"creator":"dernst","access_level":"open_access","file_name":"2020_PMLR_Konstantinov.pdf"}],"ec_funded":1,"date_published":"2020-07-12T00:00:00Z","month":"07","date_updated":"2026-04-07T14:19:48Z","citation":{"apa":"Konstantinov, N. H., Frantar, E., Alistarh, D.-A., &#38; Lampert, C. (2020). On the sample complexity of adversarial multi-source PAC learning. In <i>Proceedings of the 37th International Conference on Machine Learning</i> (Vol. 119, pp. 5416–5425). Online: ML Research Press.","short":"N.H. Konstantinov, E. Frantar, D.-A. Alistarh, C. Lampert, in:, Proceedings of the 37th International Conference on Machine Learning, ML Research Press, 2020, pp. 5416–5425.","ieee":"N. H. Konstantinov, E. Frantar, D.-A. Alistarh, and C. Lampert, “On the sample complexity of adversarial multi-source PAC learning,” in <i>Proceedings of the 37th International Conference on Machine Learning</i>, Online, 2020, vol. 119, pp. 5416–5425.","chicago":"Konstantinov, Nikola H, Elias Frantar, Dan-Adrian Alistarh, and Christoph Lampert. “On the Sample Complexity of Adversarial Multi-Source PAC Learning.” In <i>Proceedings of the 37th International Conference on Machine Learning</i>, 119:5416–25. ML Research Press, 2020.","ista":"Konstantinov NH, Frantar E, Alistarh D-A, Lampert C. 2020. On the sample complexity of adversarial multi-source PAC learning. Proceedings of the 37th International Conference on Machine Learning. ICML: International Conference on Machine Learning vol. 119, 5416–5425.","mla":"Konstantinov, Nikola H., et al. “On the Sample Complexity of Adversarial Multi-Source PAC Learning.” <i>Proceedings of the 37th International Conference on Machine Learning</i>, vol. 119, ML Research Press, 2020, pp. 5416–25.","ama":"Konstantinov NH, Frantar E, Alistarh D-A, Lampert C. On the sample complexity of adversarial multi-source PAC learning. In: <i>Proceedings of the 37th International Conference on Machine Learning</i>. Vol 119. ML Research Press; 2020:5416-5425."},"day":"12","language":[{"iso":"eng"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","_id":"8724","publication":"Proceedings of the 37th International Conference on Machine Learning","page":"5416-5425","publication_identifier":{"issn":["2640-3498"]},"volume":119,"acknowledged_ssus":[{"_id":"ScienComp"}],"has_accepted_license":"1","type":"conference","department":[{"_id":"DaAl"},{"_id":"ChLa"}],"intvolume":"       119","file_date_updated":"2021-02-15T09:00:01Z","scopus_import":"1","status":"public","related_material":{"link":[{"url":"http://proceedings.mlr.press/v119/konstantinov20a/konstantinov20a-supp.pdf","relation":"supplementary_material"}],"record":[{"id":"10799","status":"public","relation":"dissertation_contains"}]},"ddc":["000"],"corr_author":"1","oa_version":"Published Version","author":[{"last_name":"Konstantinov","full_name":"Konstantinov, Nikola H","first_name":"Nikola H","orcid":"0009-0009-5204-7621","id":"4B9D76E4-F248-11E8-B48F-1D18A9856A87"},{"id":"09a8f98d-ec99-11ea-ae11-c063a7b7fe5f","full_name":"Frantar, Elias","last_name":"Frantar","first_name":"Elias"},{"first_name":"Dan-Adrian","last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0003-3650-940X"},{"first_name":"Christoph","full_name":"Lampert, Christoph","last_name":"Lampert","orcid":"0000-0001-8622-7887","id":"40C20FD2-F248-11E8-B48F-1D18A9856A87"}],"acknowledgement":"Dan Alistarh is supported in part by the European Research Council (ERC) under the European Union’s Horizon 2020 research and innovation programme (grant agreement No 805223 ScaleML). This research was supported by the Scientific Service Units (SSU) of IST Austria through resources provided by Scientific Computing (SciComp).","quality_controlled":"1","oa":1,"project":[{"_id":"268A44D6-B435-11E9-9278-68D0E5697425","name":"Elastic Coordination for Scalable Machine Learning","grant_number":"805223","call_identifier":"H2020"}],"date_created":"2020-11-05T15:25:58Z","external_id":{"arxiv":["2002.10384"]},"arxiv":1,"title":"On the sample complexity of adversarial multi-source PAC learning","publisher":"ML Research Press","conference":{"start_date":"2020-07-12","end_date":"2020-07-18","location":"Online","name":"ICML: International Conference on Machine Learning"}}]
