[{"acknowledged_ssus":[{"_id":"ScienComp"}],"keyword":["machine learning","high-dimensional statistics","deep learning theory","privacy","memorization","robustness"],"das_tickbox":"0","corr_author":"1","degree_awarded":"PhD","doi_confirm":"1","author":[{"first_name":"Simone","full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","last_name":"Bombari"}],"abstract":[{"lang":"eng","text":"Artificial intelligence and machine learning have undergone an unprecedented evolution in the past decade, motivating a research effort toward a theory able to capture the qualitative behavior of large-scale neural systems. A central puzzle has been the clear benefit of scaling architecture size and overfitting the training set in supervised learning tasks. This evidence, in apparent contradiction with classical statistical learning theory, pushed researchers to develop a new theory capturing the interplay between the algorithmic and architectural bias of training and the specific target function, differently from previous methods rooted in uniform stability.\r\nThis approach has enabled a grounded understanding of novel learning regimes, typically through formal limits where the number of training samples $n$, data dimensions $d$, and model parameters $p$ grow to infinity at different rates. \\\\\r\nIn this thesis, we follow this approach, focusing on the trustworthiness of high-dimensional models: properties that are difficult to control during training or deployment and often emerge under unpredictable or adversarial conditions. In such settings, it is crucial to formally ensure a priori the reliability of machine learning systems.\r\nFirst, we study data memorization, both as label fitting and as the storage of private information about training samples in trained parameters. We prove that $p = \\Omega(n)$ parameters are sufficient for a deep neural network to memorize a generic set of labels, and for a model to memorize spurious features across training data. We then give evidence that $p = \\Omega(dn)$ parameters are instead necessary for an adversary to reconstruct the full training set from the trained parameters.\r\nSecond, we study robustness, both to adversarial perturbations and to distribution shift. We first prove that $p = \\Omega(dn)$ parameters can be sufficient for a class of neural networks to overfit the training data while guaranteeing robustness to adversarial perturbations. Then, we focus on spurious correlations learning in high-dimensional regression, studying the effect of the ridge regularization parameter in the proportional regime $n = \\Theta(d)$, and connecting it via an equivalence argument to the role of over-parameterization $p = \\Omega(n)$ in neural networks. We also investigate the architectural bias of attention-based networks, showing that they are sensitive to the replacement of individual words in an embedded sentence, allowing them to generalize on sentences where the contextual meaning depends on one or few words.\r\nFinally, we study differentially private optimization in high-dimensional regimes. We prove that standard private gradient methods do not suffer in the over-parameterized regime $p = \\Omega(n)$, challenging the current wisdom based on stability-derived generalization bounds. We then consider linear regression in the proportional regime $n = \\Theta(d)$, showing that standard private gradient descent can achieve optimal rates under appropriate hyper-parameter scaling, such as sufficiently small gradient clipping constants, whose role is still debated in practice."}],"related_material":{"record":[{"status":"public","relation":"part_of_dissertation","id":"12537"},{"id":"18972","relation":"part_of_dissertation","status":"public"},{"status":"public","relation":"part_of_dissertation","id":"18973"},{"status":"public","id":"21324","relation":"part_of_dissertation"},{"relation":"part_of_dissertation","id":"22894","status":"public"},{"relation":"part_of_dissertation","id":"19627","status":"public"},{"status":"public","id":"12859","relation":"part_of_dissertation"}]},"publication_identifier":{"isbn":["978-3-99078-091-6"],"issn":["2663-337X"]},"day":"08","fulldoi":"https://doi.org/10.15479/AT-ISTA-22857","ddc":["519"],"file_date_updated":"2026-09-10T10:03:48Z","doi":"10.15479/AT-ISTA-22857","OA_place":"publisher","publisher":"Institute of Science and Technology Austria","article_processing_charge":"No","oa":1,"title":"Trustworthy machine learning in high dimensions","alternative_title":["ISTA Thesis"],"month":"09","_id":"22857","status":"public","year":"2026","citation":{"chicago":"Bombari, Simone. “Trustworthy Machine Learning in High Dimensions.” Institute of Science and Technology Austria, 2026. <a href=\"https://doi.org/10.15479/AT-ISTA-22857\">https://doi.org/10.15479/AT-ISTA-22857</a>.","ieee":"S. Bombari, “Trustworthy machine learning in high dimensions,” Institute of Science and Technology Austria, 2026.","short":"S. Bombari, Trustworthy Machine Learning in High Dimensions, Institute of Science and Technology Austria, 2026.","ama":"Bombari S. Trustworthy machine learning in high dimensions. 2026. doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22857\">10.15479/AT-ISTA-22857</a>","ista":"Bombari S. 2026. Trustworthy machine learning in high dimensions. Institute of Science and Technology Austria.","apa":"Bombari, S. (2026). <i>Trustworthy machine learning in high dimensions</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/AT-ISTA-22857\">https://doi.org/10.15479/AT-ISTA-22857</a>","mla":"Bombari, Simone. <i>Trustworthy Machine Learning in High Dimensions</i>. Institute of Science and Technology Austria, 2026, doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22857\">10.15479/AT-ISTA-22857</a>."},"page":"446","acknowledgement":"This project was partially supported by the 2019 Lopez-Loreta prize,\r\nthe European Union (ERC, INF2\r\n, project number 101161364), the Austrian Science Fund\r\n(FWF) 10.55776/COE12, and a Google PhD fellowship in machine intelligence. Furthermore,\r\nthe candidate acknowledges the support from the Scientific Service Units of the Institute of\r\nScience and Technology Austria through resources provided by Scientific Computing.","project":[{"_id":"92099302-16d5-11f0-9cad-f9a785f54fbd","name":"Trustworthy Deep Learning Theory: Private Over-Parameterized Models and Robust LLMs"},{"name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","grant_number":"101161364","_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf"},{"_id":"74caaef7-b034-11f1-8f2d-e0e993bb422e","name":"Bilateral Artificial Intelligence (Mondelli)","grant_number":"COE12"}],"supervisor":[{"last_name":"Mondelli","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","orcid":"0000-0002-3242-7020","first_name":"Marco"}],"supplementarymaterial":"no","date_updated":"2026-09-21T13:07:01Z","department":[{"_id":"GradSch"},{"_id":"MaMo"}],"publication_status":"published","file":[{"file_name":"Thesis copy.zip","content_type":"application/zip","date_updated":"2026-09-08T13:28:38Z","checksum":"8eab99d6dc6e4476826bdfc7c00fdebb","relation":"source_file","file_id":"22859","access_level":"closed","date_created":"2026-09-08T13:28:38Z","creator":"sbombari","file_size":99154325},{"file_size":12856211,"creator":"sbombari","date_created":"2026-09-10T10:03:48Z","access_level":"open_access","file_id":"22898","relation":"main_file","checksum":"007033bafe4622ff4c2e758301354df1","date_updated":"2026-09-10T10:03:48Z","success":1,"content_type":"application/pdf","file_name":"2026_Bombari_Simone_Thesis.pdf"}],"language":[{"iso":"eng"}],"date_created":"2026-09-08T13:40:08Z","user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","oa_version":"Published Version","type":"dissertation","has_accepted_license":"1","date_published":"2026-09-08T00:00:00Z","researchdata_availability":"no"},{"external_id":{"arxiv":["2509.22214"]},"page":"145275-145314","citation":{"chicago":"Iurada, Leonardo, Simone Bombari, Tatiana Tommasi, and Marco Mondelli. “A Law of Data Reconstruction for Random Features (and Beyond).” In <i>14th International Conference on Learning Representations</i>, 2026:145275–314. OpenReview, 2026.","short":"L. Iurada, S. Bombari, T. Tommasi, M. Mondelli, in:, 14th International Conference on Learning Representations, OpenReview, 2026, pp. 145275–145314.","ieee":"L. Iurada, S. Bombari, T. Tommasi, and M. Mondelli, “A law of data reconstruction for random features (and beyond),” in <i>14th International Conference on Learning Representations</i>, Rio de Janeiro, Brazil, 2026, vol. 2026, pp. 145275–145314.","apa":"Iurada, L., Bombari, S., Tommasi, T., &#38; Mondelli, M. (2026). A law of data reconstruction for random features (and beyond). In <i>14th International Conference on Learning Representations</i> (Vol. 2026, pp. 145275–145314). Rio de Janeiro, Brazil: OpenReview.","mla":"Iurada, Leonardo, et al. “A Law of Data Reconstruction for Random Features (and Beyond).” <i>14th International Conference on Learning Representations</i>, vol. 2026, OpenReview, 2026, pp. 145275–314.","ama":"Iurada L, Bombari S, Tommasi T, Mondelli M. A law of data reconstruction for random features (and beyond). In: <i>14th International Conference on Learning Representations</i>. Vol 2026. OpenReview; 2026:145275-145314.","ista":"Iurada L, Bombari S, Tommasi T, Mondelli M. 2026. A law of data reconstruction for random features (and beyond). 14th International Conference on Learning Representations. ICLR: International Conference on Learning Representations  vol. 2026, 145275–145314."},"project":[{"_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf","name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","grant_number":"101161364"},{"_id":"92099302-16d5-11f0-9cad-f9a785f54fbd","name":"Trustworthy Deep Learning Theory: Private Over-Parameterized Models and Robust LLMs"}],"volume":2026,"acknowledgement":"M.M. is funded by the European Union (ERC, INF2\r\n, project number 101161364). S.B. was supported\r\nby a Google PhD fellowship. L.I. acknowledges the grant received from the European Union NextGenerationEU (Piano Nazionale di Ripresa E Resilienza (PNRR)) DM 351 on Trustworthy AI. T.T. &\r\nL.I. acknowledge the EU project ELSA - European Lighthouse on Secure and Safe AI. This study was\r\ncarried out within the FAIR - Future Artificial Intelligence Research and received funding from the\r\nEuropean Union Next-GenerationEU (PIANO NAZIONALE DI RIPRESA E RESILIENZA (PNRR)\r\n– MISSIONE 4 COMPONENTE 2, INVESTIMENTO 1.3 – D.D. 1555 11/10/2022, PE00000013).\r\nThis manuscript reflects only the authors’ views and opinions, neither the European Union nor the\r\nEuropean Commission can be considered responsible for them. The authors would like to thank\r\nYizhe Zhu for helpful discussions.","file":[{"date_created":"2026-09-11T12:07:48Z","creator":"cchlebak","file_size":7869510,"access_level":"open_access","date_updated":"2026-09-11T12:07:48Z","checksum":"a3dcba6649b7496124fa0ae68bf2c4cb","relation":"main_file","file_id":"22907","file_name":"2026_ICLR_Iurada.pdf","content_type":"application/pdf","success":1}],"tmp":{"image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","short":"CC BY (4.0)"},"department":[{"_id":"GradSch"},{"_id":"MaMo"}],"publication_status":"published","date_updated":"2026-09-21T13:07:01Z","date_published":"2026-01-26T00:00:00Z","has_accepted_license":"1","type":"conference","arxiv":1,"quality_controlled":"1","intvolume":"      2026","user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","date_created":"2026-09-09T13:31:46Z","oa_version":"Published Version","language":[{"iso":"eng"}],"corr_author":"1","das_tickbox":"0","OA_type":"gold","related_material":{"record":[{"id":"22857","relation":"dissertation_contains","status":"public"}]},"abstract":[{"text":"Large-scale deep learning models are known to memorize parts of the training\r\nset. In machine learning theory, memorization is often framed as interpolation or\r\nlabel fitting, and classical results show that this can be achieved when the number\r\nof parameters p in the model is larger than the number of training samples n. In\r\nthis work, we consider memorization from the perspective of data reconstruction,\r\ndemonstrating that this can be achieved when p is larger than dn, where d is\r\nthe dimensionality of the data. More specifically, we show that, in the random\r\nfeatures model, when p ≫ dn, the subspace spanned by the training samples in\r\nfeature space gives sufficient information to identify the individual samples in input\r\nspace. Our analysis suggests an optimization method to reconstruct the dataset\r\nfrom the model parameters, and we demonstrate that this method performs well on\r\nvarious architectures (random features, two-layer fully-connected and deep residual\r\nnetworks). Our results reveal a law of data reconstruction, according to which the\r\nentire training dataset can be recovered as p exceeds the threshold dn.\r\n","lang":"eng"}],"author":[{"last_name":"Iurada","full_name":"Iurada, Leonardo","first_name":"Leonardo"},{"full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","first_name":"Simone","last_name":"Bombari"},{"last_name":"Tommasi","first_name":"Tatiana","full_name":"Tommasi, Tatiana"},{"last_name":"Mondelli","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"publisher":"OpenReview","file_date_updated":"2026-09-11T12:07:48Z","OA_place":"publisher","ddc":["000"],"day":"26","publication_identifier":{"isbn":["9798331339678"]},"year":"2026","_id":"22894","status":"public","month":"01","conference":{"location":"Rio de Janeiro, Brazil","name":"ICLR: International Conference on Learning Representations ","end_date":"2026-04-27","start_date":"2026-04-23"},"publication":"14th International Conference on Learning Representations","title":"A law of data reconstruction for random features (and beyond)","oa":1,"article_processing_charge":"No"},{"type":"journal_article","has_accepted_license":"1","date_published":"2025-04-15T00:00:00Z","language":[{"iso":"eng"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","date_created":"2025-04-27T22:02:13Z","oa_version":"Published Version","quality_controlled":"1","intvolume":"       122","arxiv":1,"tmp":{"image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","short":"CC BY (4.0)"},"file":[{"access_level":"open_access","file_size":2328320,"creator":"dernst","date_created":"2025-05-05T07:27:54Z","success":1,"content_type":"application/pdf","file_name":"2025_PNAS_Bombari.pdf","file_id":"19648","checksum":"1ac6f78e368d35a0cafb4d2d9bd63443","relation":"main_file","date_updated":"2025-05-05T07:27:54Z"}],"article_type":"original","date_updated":"2026-09-21T13:07:01Z","department":[{"_id":"MaMo"}],"publication_status":"published","acknowledgement":"This research was funded in whole, or in part, by the Austrian Science Fund (FWF) Grant number COE 12. For the purpose of open access, the author has applied a CC BY public copyright license to any Author Accepted Manuscript version arising from this submission. The authors were also supported by the 2019 Lopez-Loreta prize, and Simone Bombari was supported by a Google PhD fellowship. We thank Diyuan Wu, Edwige Cyffers, Francesco Pedrotti, Inbar Seroussi, Nikita P. Kalinin, Pietro Pelliconi, Roodabeh Safavi, Yizhe Zhu, and Zhichao Wang for helpful discussions.","volume":122,"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"},{"name":"Trustworthy Deep Learning Theory: Private Over-Parameterized Models and Robust LLMs","_id":"92099302-16d5-11f0-9cad-f9a785f54fbd"},{"grant_number":"COE12","name":"Bilateral Artificial Intelligence (Mondelli)","_id":"74caaef7-b034-11f1-8f2d-e0e993bb422e"}],"citation":{"apa":"Bombari, S., &#38; Mondelli, M. (2025). Privacy for free in the overparameterized regime. <i>Proceedings of the National Academy of Sciences</i>. National Academy of Sciences. <a href=\"https://doi.org/10.1073/pnas.2423072122\">https://doi.org/10.1073/pnas.2423072122</a>","mla":"Bombari, Simone, and Marco Mondelli. “Privacy for Free in the Overparameterized Regime.” <i>Proceedings of the National Academy of Sciences</i>, vol. 122, no. 15, e2423072122, National Academy of Sciences, 2025, doi:<a href=\"https://doi.org/10.1073/pnas.2423072122\">10.1073/pnas.2423072122</a>.","ista":"Bombari S, Mondelli M. 2025. Privacy for free in the overparameterized regime. Proceedings of the National Academy of Sciences. 122(15), e2423072122.","ama":"Bombari S, Mondelli M. Privacy for free in the overparameterized regime. <i>Proceedings of the National Academy of Sciences</i>. 2025;122(15). doi:<a href=\"https://doi.org/10.1073/pnas.2423072122\">10.1073/pnas.2423072122</a>","short":"S. Bombari, M. Mondelli, Proceedings of the National Academy of Sciences 122 (2025).","ieee":"S. Bombari and M. Mondelli, “Privacy for free in the overparameterized regime,” <i>Proceedings of the National Academy of Sciences</i>, vol. 122, no. 15. National Academy of Sciences, 2025.","chicago":"Bombari, Simone, and Marco Mondelli. “Privacy for Free in the Overparameterized Regime.” <i>Proceedings of the National Academy of Sciences</i>. National Academy of Sciences, 2025. <a href=\"https://doi.org/10.1073/pnas.2423072122\">https://doi.org/10.1073/pnas.2423072122</a>."},"external_id":{"isi":["001471214000001"],"pmid":["40215275"],"arxiv":["2410.14787"]},"issue":"15","publication":"Proceedings of the National Academy of Sciences","_id":"19627","status":"public","month":"04","year":"2025","article_processing_charge":"Yes (in subscription journal)","oa":1,"scopus_import":"1","title":"Privacy for free in the overparameterized regime","ddc":["000"],"fulldoi":"https://doi.org/10.1073/pnas.2423072122","doi":"10.1073/pnas.2423072122","file_date_updated":"2025-05-05T07:27:54Z","OA_place":"publisher","publisher":"National Academy of Sciences","publication_identifier":{"issn":["0027-8424"],"eissn":["1091-6490"]},"APC_amount":"2754,32 EUR","day":"15","abstract":[{"lang":"eng","text":"Differentially private gradient descent (DP-GD) is a popular algorithm to train deep learning models with provable guarantees on the privacy of the training data. In the last decade, the problem of understanding its performance cost with respect to standard GD has received remarkable attention from the research community, which formally derived upper bounds on the excess population risk  RP  in different learning settings. However, existing bounds typically degrade with over-parameterization, i.e., as the number of parameters  p  gets larger than the number of training samples  n  -- a regime which is ubiquitous in current deep-learning practice. As a result, the lack of theoretical insights leaves practitioners without clear guidance, leading some to reduce the effective number of trainable parameters to improve performance, while others use larger models to achieve better results through scale. In this work, we show that in the popular random features model with quadratic loss, for any sufficiently large  p , privacy can be obtained for free, i.e.,  |RP|=o(1) , not only when the privacy parameter  ε  has constant order, but also in the strongly private setting  ε=o(1) . This challenges the common wisdom that over-parameterization inherently hinders performance in private learning."}],"related_material":{"record":[{"status":"public","relation":"dissertation_contains","id":"22857"}]},"pmid":1,"author":[{"first_name":"Simone","full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","last_name":"Bombari"},{"id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco","first_name":"Marco","orcid":"0000-0002-3242-7020","last_name":"Mondelli"}],"OA_type":"hybrid","corr_author":"1","isi":1,"article_number":"e2423072122"},{"OA_place":"publisher","file_date_updated":"2026-02-19T08:04:38Z","publisher":"ML Research Press","ddc":["000"],"day":"30","publication_identifier":{"eissn":["2640-3498"]},"_id":"21324","status":"public","month":"07","year":"2025","publication":"Proceedings of the 42nd International Conference on Machine Learning","alternative_title":["PMLR"],"conference":{"start_date":"2025-07-13","location":"Vancouver, Canada","end_date":"2025-07-19","name":"ICML: International Conference on Machine Learning"},"title":"Spurious correlations in high dimensional regression: The roles of regularization, simplicity bias and over-parameterization","article_processing_charge":"No","oa":1,"corr_author":"1","OA_type":"gold","related_material":{"record":[{"status":"public","relation":"dissertation_contains","id":"22857"}]},"abstract":[{"text":"Learning models have been shown to rely on spurious correlations between non-predictive features and the associated labels in the training data, with negative implications on robustness, bias and fairness. In this work, we provide a statistical characterization of this phenomenon for high-dimensional regression, when the data contains a predictive core feature x and a spurious feature y. Specifically, we quantify the amount of spurious correlations C learned via linear regression, in terms of the data covariance and the strength λ of the ridge regularization. As a consequence, we first capture the simplicity of y through the spectrum of its covariance, and its correlation with x through the Schur complement of the full data covariance. Next, we prove a trade-off between C and the in-distribution test loss L, by showing that the value of λ that minimizes L lies in an interval where C is increasing. Finally, we investigate the effects of over-parameterization via the random features model, by showing its equivalence to regularized linear regression. Our theoretical results are supported by numerical experiments on Gaussian, Color-MNIST, and CIFAR-10 datasets.","lang":"eng"}],"author":[{"first_name":"Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone","last_name":"Bombari"},{"last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"}],"tmp":{"image":"/images/cc_by.png","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","short":"CC BY (4.0)"},"file":[{"file_size":887526,"creator":"dernst","date_created":"2026-02-19T08:04:38Z","access_level":"open_access","file_id":"21335","relation":"main_file","checksum":"d4ba4f7717b362ca38878f45e57bd643","date_updated":"2026-02-19T08:04:38Z","success":1,"content_type":"application/pdf","file_name":"2025_ICML_Bombari.pdf"}],"department":[{"_id":"MaMo"}],"publication_status":"published","date_updated":"2026-09-21T13:07:01Z","date_published":"2025-07-30T00:00:00Z","type":"conference","has_accepted_license":"1","arxiv":1,"intvolume":"       267","quality_controlled":"1","language":[{"iso":"eng"}],"oa_version":"Published Version","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","date_created":"2026-02-18T11:58:00Z","page":"4839-4873","citation":{"chicago":"Bombari, Simone, and Marco Mondelli. “Spurious Correlations in High Dimensional Regression: The Roles of Regularization, Simplicity Bias and over-Parameterization.” In <i>Proceedings of the 42nd International Conference on Machine Learning</i>, 267:4839–73. ML Research Press, 2025.","ista":"Bombari S, Mondelli M. 2025. Spurious correlations in high dimensional regression: The roles of regularization, simplicity bias and over-parameterization. Proceedings of the 42nd International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 267, 4839–4873.","ama":"Bombari S, Mondelli M. Spurious correlations in high dimensional regression: The roles of regularization, simplicity bias and over-parameterization. In: <i>Proceedings of the 42nd International Conference on Machine Learning</i>. Vol 267. ML Research Press; 2025:4839-4873.","mla":"Bombari, Simone, and Marco Mondelli. “Spurious Correlations in High Dimensional Regression: The Roles of Regularization, Simplicity Bias and over-Parameterization.” <i>Proceedings of the 42nd International Conference on Machine Learning</i>, vol. 267, ML Research Press, 2025, pp. 4839–73.","apa":"Bombari, S., &#38; Mondelli, M. (2025). Spurious correlations in high dimensional regression: The roles of regularization, simplicity bias and over-parameterization. In <i>Proceedings of the 42nd International Conference on Machine Learning</i> (Vol. 267, pp. 4839–4873). Vancouver, Canada: ML Research Press.","ieee":"S. Bombari and M. Mondelli, “Spurious correlations in high dimensional regression: The roles of regularization, simplicity bias and over-parameterization,” in <i>Proceedings of the 42nd International Conference on Machine Learning</i>, Vancouver, Canada, 2025, vol. 267, pp. 4839–4873.","short":"S. Bombari, M. Mondelli, in:, Proceedings of the 42nd International Conference on Machine Learning, ML Research Press, 2025, pp. 4839–4873."},"external_id":{"arxiv":["2502.01347"]},"project":[{"_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf","name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","grant_number":"101161364"},{"_id":"92099302-16d5-11f0-9cad-f9a785f54fbd","name":"Trustworthy Deep Learning Theory: Private Over-Parameterized Models and Robust LLMs"}],"volume":267,"acknowledgement":"Marco Mondelli is funded by the European Union (ERC, INF2, project number 101161364). Views and opinions expressed are however those of the author(s) only and do not necessarily reflect those of the European Union or the European Research Council Executive Agency. Neither the European Union nor the granting authority can be held responsible for them. Simone Bombari is supported by a Google PhD fellowship. The authors would like to thank GuanWen Qiu for helpful discussions."},{"citation":{"short":"S. Bombari, M. Mondelli, in:, 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 4300–4328.","ieee":"S. Bombari and M. Mondelli, “Towards understanding the word sensitivity of attention layers: A study via random features,” in <i>41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 4300–4328.","mla":"Bombari, Simone, and Marco Mondelli. “Towards Understanding the Word Sensitivity of Attention Layers: A Study via Random Features.” <i>41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 4300–28.","apa":"Bombari, S., &#38; Mondelli, M. (2024). Towards understanding the word sensitivity of attention layers: A study via random features. In <i>41st International Conference on Machine Learning</i> (Vol. 235, pp. 4300–4328). Vienna, Austria: ML Research Press.","ama":"Bombari S, Mondelli M. Towards understanding the word sensitivity of attention layers: A study via random features. In: <i>41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:4300-4328.","ista":"Bombari S, Mondelli M. 2024. Towards understanding the word sensitivity of attention layers: A study via random features. 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 4300–4328.","chicago":"Bombari, Simone, and Marco Mondelli. “Towards Understanding the Word Sensitivity of Attention Layers: A Study via Random Features.” In <i>41st International Conference on Machine Learning</i>, 235:4300–4328. ML Research Press, 2024."},"page":"4300-4328","external_id":{"arxiv":["2402.02969"]},"main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2402.02969","open_access":"1"}],"acknowledgement":"The authors were partially supported by the 2019 LopezLoreta prize, and they would like to thank Mohammad Hossein Amani, Lorenzo Beretta, and Clement Rebuffel for helpful discussions.","volume":235,"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"date_updated":"2026-09-21T13:07:01Z","publication_status":"published","department":[{"_id":"MaMo"}],"type":"conference","date_published":"2024-07-30T00:00:00Z","language":[{"iso":"eng"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","date_created":"2025-01-30T07:35:49Z","oa_version":"Preprint","arxiv":1,"quality_controlled":"1","intvolume":"       235","OA_type":"green","corr_author":"1","abstract":[{"text":"Understanding the reasons behind the exceptional success of transformers requires a better analysis of why attention layers are suitable for NLP tasks. In particular, such tasks require predictive models to capture contextual meaning which often depends on one or few words, even if the sentence is long. Our work studies this key property, dubbed word sensitivity (WS), in the prototypical setting of random features. We show that attention layers enjoy high WS, namely, there exists a vector in the space of embeddings that largely perturbs the random attention features map. The argument critically exploits the role of the softmax in the attention layer, highlighting its benefit compared to other activations (e.g., ReLU). In contrast, the WS of standard random features is of order 1/n−−√, n being the number of words in the textual sample, and thus it decays with the length of the context. We then translate these results on the word sensitivity into generalization bounds: due to their low WS, random features provably cannot learn to distinguish between two sentences that differ only in a single word; in contrast, due to their high WS, random attention features have higher generalization capabilities. We validate our theoretical results with experimental evidence over the BERT-Base word embeddings of the imdb review dataset.","lang":"eng"}],"related_material":{"record":[{"id":"22857","relation":"dissertation_contains","status":"public"}]},"author":[{"last_name":"Bombari","first_name":"Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone"},{"last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"}],"OA_place":"repository","publisher":"ML Research Press","publication_identifier":{"eissn":["2640-3498"]},"day":"30","alternative_title":["PMLR"],"publication":"41st International Conference on Machine Learning","conference":{"location":"Vienna, Austria","name":"ICML: International Conference on Machine Learning","end_date":"2024-07-27","start_date":"2024-07-21"},"status":"public","_id":"18973","month":"07","year":"2024","article_processing_charge":"No","oa":1,"scopus_import":"1","title":"Towards understanding the word sensitivity of attention layers: A study via random features"},{"date_published":"2024-07-30T00:00:00Z","type":"conference","quality_controlled":"1","intvolume":"       235","arxiv":1,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","date_created":"2025-01-30T07:29:47Z","oa_version":"Preprint","language":[{"iso":"eng"}],"department":[{"_id":"MaMo"}],"publication_status":"published","date_updated":"2026-09-21T13:07:01Z","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"acknowledgement":"The authors were partially supported by the 2019 LopezLoreta prize, and they would like to thank (in alphabetical order) Grigorios Chrysos, Simone Maria Giancola, Mahyar\r\nJafari Nodeh, Christoph Lampert, Marco Miani, GuanWen Qiu, and Peter Sukenık for helpful discussions.","volume":235,"main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2305.12100"}],"external_id":{"arxiv":["2305.12100"]},"citation":{"ieee":"S. Bombari and M. Mondelli, “How spurious features are memorized: Precise analysis for random and NTK features,” in <i>41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 4267–4299.","short":"S. Bombari, M. Mondelli, in:, 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 4267–4299.","ista":"Bombari S, Mondelli M. 2024. How spurious features are memorized: Precise analysis for random and NTK features. 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 4267–4299.","ama":"Bombari S, Mondelli M. How spurious features are memorized: Precise analysis for random and NTK features. In: <i>41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:4267-4299.","apa":"Bombari, S., &#38; Mondelli, M. (2024). How spurious features are memorized: Precise analysis for random and NTK features. In <i>41st International Conference on Machine Learning</i> (Vol. 235, pp. 4267–4299). Vienna, Austria: ML Research Press.","mla":"Bombari, Simone, and Marco Mondelli. “How Spurious Features Are Memorized: Precise Analysis for Random and NTK Features.” <i>41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 4267–99.","chicago":"Bombari, Simone, and Marco Mondelli. “How Spurious Features Are Memorized: Precise Analysis for Random and NTK Features.” In <i>41st International Conference on Machine Learning</i>, 235:4267–99. ML Research Press, 2024."},"page":"4267-4299","year":"2024","_id":"18972","month":"07","status":"public","conference":{"start_date":"2024-07-21","name":"ICML: International Conference on Machine Learning","end_date":"2024-07-27","location":"Vienna, Austria"},"publication":"41st International Conference on Machine Learning","alternative_title":["PMLR"],"title":"How spurious features are memorized: Precise analysis for random and NTK features","scopus_import":"1","oa":1,"article_processing_charge":"No","publisher":"ML Research Press","OA_place":"repository","day":"30","publication_identifier":{"eissn":["2640-3498"]},"related_material":{"record":[{"relation":"dissertation_contains","id":"22857","status":"public"}]},"abstract":[{"lang":"eng","text":"Deep learning models are known to overfit and memorize spurious features in the training dataset. While numerous empirical studies have aimed at understanding this phenomenon, a rigorous theoretical framework to quantify it is still missing. In this paper, we consider spurious features that are uncorrelated with the learning task, and we provide a precise characterization of how they are memorized via two separate terms: (i) the stability of the model with respect to individual training samples, and (ii) the feature alignment between the spurious pattern and the full sample. While the first term is well established in learning theory and it is connected to the generalization error in classical work, the second one is, to the best of our knowledge, novel. Our key technical result gives a precise characterization of the feature alignment for the two prototypical settings of random features (RF) and neural tangent kernel (NTK) regression. We prove that the memorization of spurious features weakens as the generalization capability increases and, through the analysis of the feature alignment, we unveil the role of the model and of its activation function. Numerical experiments show the predictive power of our theory on standard datasets (MNIST, CIFAR-10)."}],"author":[{"last_name":"Bombari","first_name":"Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone"},{"last_name":"Mondelli","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"corr_author":"1","OA_type":"green"},{"citation":{"chicago":"Bombari, Simone, Shayan Kiyani, and Marco Mondelli. “Beyond the Universal Law of Robustness: Sharper Laws for Random Features and Neural Tangent Kernels.” In <i>Proceedings of the 40th International Conference on Machine Learning</i>, 202:2738–76. ML Research Press, 2023.","ama":"Bombari S, Kiyani S, Mondelli M. Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. In: <i>Proceedings of the 40th International Conference on Machine Learning</i>. Vol 202. ML Research Press; 2023:2738-2776.","ista":"Bombari S, Kiyani S, Mondelli M. 2023. Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. Proceedings of the 40th International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 202, 2738–2776.","apa":"Bombari, S., Kiyani, S., &#38; Mondelli, M. (2023). Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. In <i>Proceedings of the 40th International Conference on Machine Learning</i> (Vol. 202, pp. 2738–2776). Honolulu, HI, United States: ML Research Press.","mla":"Bombari, Simone, et al. “Beyond the Universal Law of Robustness: Sharper Laws for Random Features and Neural Tangent Kernels.” <i>Proceedings of the 40th International Conference on Machine Learning</i>, vol. 202, ML Research Press, 2023, pp. 2738–76.","ieee":"S. Bombari, S. Kiyani, and M. Mondelli, “Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels,” in <i>Proceedings of the 40th International Conference on Machine Learning</i>, Honolulu, HI, United States, 2023, vol. 202, pp. 2738–2776.","short":"S. Bombari, S. Kiyani, M. Mondelli, in:, Proceedings of the 40th International Conference on Machine Learning, ML Research Press, 2023, pp. 2738–2776."},"page":"2738-2776","external_id":{"arxiv":["2302.01629"]},"main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2302.01629"}],"acknowledgement":"Simone Bombari and Marco Mondelli were partially supported by the 2019 Lopez-Loreta prize, and\r\nthe authors would like to thank Hamed Hassani for helpful discussions.\r\n","volume":202,"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"date_updated":"2026-09-21T13:07:01Z","publication_status":"published","department":[{"_id":"GradSch"},{"_id":"MaMo"}],"type":"conference","date_published":"2023-10-27T00:00:00Z","language":[{"iso":"eng"}],"oa_version":"Preprint","date_created":"2023-04-23T16:11:03Z","user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","intvolume":"       202","arxiv":1,"quality_controlled":"1","corr_author":"1","abstract":[{"lang":"eng","text":"Machine learning models are vulnerable to adversarial perturbations, and a thought-provoking paper by Bubeck and Sellke has analyzed this phenomenon through the lens of over-parameterization: interpolating smoothly the data requires significantly more parameters than simply memorizing it. However, this \"universal\" law provides only a necessary condition for robustness, and it is unable to discriminate between models. In this paper, we address these gaps by focusing on empirical risk minimization in two prototypical settings, namely, random features and the neural tangent kernel (NTK). We prove that, for random features, the model is not robust for any degree of over-parameterization, even when the necessary condition coming from the universal law of robustness is satisfied. In contrast, for even activations, the NTK model meets the universal lower bound, and it is robust as soon as the necessary condition on over-parameterization is fulfilled. This also addresses a conjecture in prior work by Bubeck, Li and Nagaraj. Our analysis decouples the effect of the kernel of the model from an \"interaction matrix\", which describes the interaction with the test data and captures the effect of the activation. Our theoretical results are corroborated by numerical evidence on both synthetic and standard datasets (MNIST, CIFAR-10)."}],"related_material":{"record":[{"relation":"dissertation_contains","id":"22857","status":"public"}],"link":[{"relation":"software","url":"https://github.com/simone-bombari/beyond-universal-robustness"}]},"author":[{"full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","first_name":"Simone","last_name":"Bombari"},{"full_name":"Kiyani, Shayan","id":"f5a2b424-e339-11ed-8435-ff3b4fe70cf8","first_name":"Shayan","last_name":"Kiyani"},{"orcid":"0000-0002-3242-7020","first_name":"Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco","last_name":"Mondelli"}],"publisher":"ML Research Press","day":"27","alternative_title":["PMLR"],"publication":"Proceedings of the 40th International Conference on Machine Learning","conference":{"end_date":"2023-07-29","name":"ICML: International Conference on Machine Learning","location":"Honolulu, HI, United States","start_date":"2023-07-23"},"_id":"12859","month":"10","status":"public","year":"2023","article_processing_charge":"No","oa":1,"title":"Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels"},{"date_updated":"2025-09-10T09:53:31Z","publication_status":"published","department":[{"_id":"MaMo"}],"article_type":"original","date_created":"2023-02-10T13:47:56Z","user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","oa_version":"Preprint","language":[{"iso":"eng"}],"arxiv":1,"quality_controlled":"1","type":"journal_article","date_published":"2022-11-16T00:00:00Z","external_id":{"arxiv":["2205.08199"],"isi":["000904341100099"]},"citation":{"chicago":"Amani, Mohammad Hossein, Simone Bombari, Marco Mondelli, Rattana Pukdee, and Stefano Rini. “Sharp Asymptotics on the Compression of Two-Layer Neural Networks.” <i>IEEE Information Theory Workshop</i>. IEEE, 2022. <a href=\"https://doi.org/10.1109/ITW54588.2022.9965870\">https://doi.org/10.1109/ITW54588.2022.9965870</a>.","ista":"Amani MH, Bombari S, Mondelli M, Pukdee R, Rini S. 2022. Sharp asymptotics on the compression of two-layer neural networks. IEEE Information Theory Workshop., 588–593.","ama":"Amani MH, Bombari S, Mondelli M, Pukdee R, Rini S. Sharp asymptotics on the compression of two-layer neural networks. <i>IEEE Information Theory Workshop</i>. 2022:588-593. doi:<a href=\"https://doi.org/10.1109/ITW54588.2022.9965870\">10.1109/ITW54588.2022.9965870</a>","apa":"Amani, M. H., Bombari, S., Mondelli, M., Pukdee, R., &#38; Rini, S. (2022). Sharp asymptotics on the compression of two-layer neural networks. <i>IEEE Information Theory Workshop</i>. Mumbai, India: IEEE. <a href=\"https://doi.org/10.1109/ITW54588.2022.9965870\">https://doi.org/10.1109/ITW54588.2022.9965870</a>","mla":"Amani, Mohammad Hossein, et al. “Sharp Asymptotics on the Compression of Two-Layer Neural Networks.” <i>IEEE Information Theory Workshop</i>, IEEE, 2022, pp. 588–93, doi:<a href=\"https://doi.org/10.1109/ITW54588.2022.9965870\">10.1109/ITW54588.2022.9965870</a>.","ieee":"M. H. Amani, S. Bombari, M. Mondelli, R. Pukdee, and S. Rini, “Sharp asymptotics on the compression of two-layer neural networks,” <i>IEEE Information Theory Workshop</i>. IEEE, pp. 588–593, 2022.","short":"M.H. Amani, S. Bombari, M. Mondelli, R. Pukdee, S. Rini, IEEE Information Theory Workshop (2022) 588–593."},"page":"588-593","main_file_link":[{"url":" https://doi.org/10.48550/arXiv.2205.08199","open_access":"1"}],"publication_identifier":{"isbn":["9781665483414"]},"day":"16","fulldoi":"https://doi.org/10.1109/ITW54588.2022.9965870","publisher":"IEEE","doi":"10.1109/ITW54588.2022.9965870","oa":1,"article_processing_charge":"No","title":"Sharp asymptotics on the compression of two-layer neural networks","scopus_import":"1","conference":{"location":"Mumbai, India","name":"ITW: Information Theory Workshop","end_date":"2022-11-09","start_date":"2022-11-01"},"publication":"IEEE Information Theory Workshop","year":"2022","_id":"12538","month":"11","status":"public","isi":1,"author":[{"full_name":"Amani, Mohammad Hossein","first_name":"Mohammad Hossein","last_name":"Amani"},{"last_name":"Bombari","first_name":"Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone"},{"last_name":"Mondelli","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","first_name":"Marco","orcid":"0000-0002-3242-7020"},{"last_name":"Pukdee","full_name":"Pukdee, Rattana","first_name":"Rattana"},{"last_name":"Rini","first_name":"Stefano","full_name":"Rini, Stefano"}],"abstract":[{"text":"In this paper, we study the compression of a target two-layer neural network with N nodes into a compressed network with M<N nodes. More precisely, we consider the setting in which the weights of the target network are i.i.d. sub-Gaussian, and we minimize the population L_2 loss between the outputs of the target and of the compressed network, under the assumption of Gaussian inputs. By using tools from high-dimensional probability, we show that this non-convex problem can be simplified when the target network is sufficiently over-parameterized, and provide the error rate of this approximation as a function of the input dimension and N. In this mean-field limit, the simplified objective, as well as the optimal weights of the compressed network, does not depend on the realization of the target network, but only on expected scaling factors. Furthermore, for networks with ReLU activation, we conjecture that the optimum of the simplified optimization problem is achieved by taking weights on the Equiangular Tight Frame (ETF), while the scaling of the weights and the orientation of the ETF depend on the parameters of the target network. Numerical evidence is provided to support this conjecture.","lang":"eng"}]},{"publication":"arXiv","type":"preprint","month":"03","_id":"12860","status":"public","date_published":"2022-03-30T00:00:00Z","year":"2022","language":[{"iso":"eng"}],"article_processing_charge":"No","oa":1,"date_created":"2023-04-23T16:11:48Z","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","oa_version":"Preprint","title":"Towards differential relational privacy and its use in question answering","arxiv":1,"fulldoi":"https://doi.org/10.48550/arXiv.2203.16701","doi":"10.48550/arXiv.2203.16701","date_updated":"2023-04-25T07:34:49Z","publication_status":"submitted","day":"30","department":[{"_id":"GradSch"},{"_id":"MaMo"}],"abstract":[{"text":"Memorization of the relation between entities in a dataset can lead to privacy issues when using a trained model for question answering. We introduce Relational Memorization (RM) to understand, quantify and control this phenomenon. While bounding general memorization can have detrimental effects on the performance of a trained model, bounding RM does not prevent effective learning. The difference is most pronounced when the data distribution is long-tailed, with many queries having only few training examples: Impeding general memorization prevents effective learning, while impeding only relational memorization still allows learning general properties of the underlying concepts. We formalize the notion of Relational Privacy (RP) and, inspired by Differential Privacy (DP), we provide a possible definition of Differential Relational Privacy (DrP). These notions can be used to describe and compute bounds on the amount of RM in a trained model. We illustrate Relational Privacy concepts in experiments with large-scale models for Question Answering.","lang":"eng"}],"author":[{"last_name":"Bombari","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone","first_name":"Simone"},{"last_name":"Achille","first_name":"Alessandro","full_name":"Achille, Alessandro"},{"first_name":"Zijian","full_name":"Wang, Zijian","last_name":"Wang"},{"last_name":"Wang","first_name":"Yu-Xiang","full_name":"Wang, Yu-Xiang"},{"first_name":"Yusheng","full_name":"Xie, Yusheng","last_name":"Xie"},{"full_name":"Singh, Kunwar Yashraj","first_name":"Kunwar Yashraj","last_name":"Singh"},{"first_name":"Srikar","full_name":"Appalaraju, Srikar","last_name":"Appalaraju"},{"full_name":"Mahadevan, Vijay","first_name":"Vijay","last_name":"Mahadevan"},{"full_name":"Soatto, Stefano","first_name":"Stefano","last_name":"Soatto"}],"citation":{"chicago":"Bombari, Simone, Alessandro Achille, Zijian Wang, Yu-Xiang Wang, Yusheng Xie, Kunwar Yashraj Singh, Srikar Appalaraju, Vijay Mahadevan, and Stefano Soatto. “Towards Differential Relational Privacy and Its Use in Question Answering.” <i>ArXiv</i>, n.d. <a href=\"https://doi.org/10.48550/arXiv.2203.16701\">https://doi.org/10.48550/arXiv.2203.16701</a>.","short":"S. Bombari, A. Achille, Z. Wang, Y.-X. Wang, Y. Xie, K.Y. Singh, S. Appalaraju, V. Mahadevan, S. Soatto, ArXiv (n.d.).","ieee":"S. Bombari <i>et al.</i>, “Towards differential relational privacy and its use in question answering,” <i>arXiv</i>. .","mla":"Bombari, Simone, et al. “Towards Differential Relational Privacy and Its Use in Question Answering.” <i>ArXiv</i>, 2203.16701, doi:<a href=\"https://doi.org/10.48550/arXiv.2203.16701\">10.48550/arXiv.2203.16701</a>.","apa":"Bombari, S., Achille, A., Wang, Z., Wang, Y.-X., Xie, Y., Singh, K. Y., … Soatto, S. (n.d.). Towards differential relational privacy and its use in question answering. <i>arXiv</i>. <a href=\"https://doi.org/10.48550/arXiv.2203.16701\">https://doi.org/10.48550/arXiv.2203.16701</a>","ista":"Bombari S, Achille A, Wang Z, Wang Y-X, Xie Y, Singh KY, Appalaraju S, Mahadevan V, Soatto S. Towards differential relational privacy and its use in question answering. arXiv, 2203.16701.","ama":"Bombari S, Achille A, Wang Z, et al. Towards differential relational privacy and its use in question answering. <i>arXiv</i>. doi:<a href=\"https://doi.org/10.48550/arXiv.2203.16701\">10.48550/arXiv.2203.16701</a>"},"external_id":{"arxiv":["2203.16701"]},"main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2203.16701","open_access":"1"}],"article_number":"2203.16701"},{"author":[{"first_name":"Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","full_name":"Bombari, Simone","last_name":"Bombari"},{"last_name":"Amani","full_name":"Amani, Mohammad Hossein","first_name":"Mohammad Hossein"},{"orcid":"0000-0002-3242-7020","first_name":"Marco","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli"}],"abstract":[{"text":"The Neural Tangent Kernel (NTK) has emerged as a powerful tool to provide memorization, optimization and generalization guarantees in deep neural networks. A line of work has studied the NTK spectrum for two-layer and deep networks with at least a layer with Ω(N) neurons, N being the number of training samples. Furthermore, there is increasing evidence suggesting that deep networks with sub-linear layer widths are powerful memorizers and optimizers, as long as the number of parameters exceeds the number of samples. Thus, a natural open question is whether the NTK is well conditioned in such a challenging sub-linear setup. In this paper, we answer this question in the affirmative. Our key technical contribution is a lower bound on the smallest NTK eigenvalue for deep networks with the minimum possible over-parameterization: the number of parameters is roughly Ω(N) and, hence, the number of neurons is as little as Ω(N−−√). To showcase the applicability of our NTK bounds, we provide two results concerning memorization capacity and optimization guarantees for gradient descent training.","lang":"eng"}],"related_material":{"record":[{"status":"public","id":"22857","relation":"dissertation_contains"}]},"OA_type":"green","corr_author":"1","article_processing_charge":"No","oa":1,"title":"Memorization and optimization in deep neural networks with minimum over-parameterization","publication":"36th Conference on Neural Information Processing Systems","alternative_title":["Advances in Neural Information Processing Systems"],"conference":{"start_date":"2022-11-28","name":"NeurIPS: Neural Information Processing Systems","end_date":"2022-12-09","location":"New Orleans, LA, United States"},"_id":"12537","status":"public","month":"07","year":"2022","publication_identifier":{"eissn":["1049-5258"],"isbn":["9781713871088"]},"day":"24","OA_place":"repository","publisher":"Neural Information Processing Systems Foundation","acknowledgement":"The authors were partially supported by the 2019 Lopez-Loreta prize, and they would like to thank\r\nQuynh Nguyen, Mahdi Soltanolkotabi and Adel Javanmard for helpful discussions.\r\n","volume":35,"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"citation":{"short":"S. Bombari, M.H. Amani, M. Mondelli, in:, 36th Conference on Neural Information Processing Systems, Neural Information Processing Systems Foundation, 2022, pp. 7628–7640.","ieee":"S. Bombari, M. H. Amani, and M. Mondelli, “Memorization and optimization in deep neural networks with minimum over-parameterization,” in <i>36th Conference on Neural Information Processing Systems</i>, New Orleans, LA, United States, 2022, vol. 35, pp. 7628–7640.","mla":"Bombari, Simone, et al. “Memorization and Optimization in Deep Neural Networks with Minimum Over-Parameterization.” <i>36th Conference on Neural Information Processing Systems</i>, vol. 35, Neural Information Processing Systems Foundation, 2022, pp. 7628–40.","apa":"Bombari, S., Amani, M. H., &#38; Mondelli, M. (2022). Memorization and optimization in deep neural networks with minimum over-parameterization. In <i>36th Conference on Neural Information Processing Systems</i> (Vol. 35, pp. 7628–7640). New Orleans, LA, United States: Neural Information Processing Systems Foundation.","ama":"Bombari S, Amani MH, Mondelli M. Memorization and optimization in deep neural networks with minimum over-parameterization. In: <i>36th Conference on Neural Information Processing Systems</i>. Vol 35. Neural Information Processing Systems Foundation; 2022:7628-7640.","ista":"Bombari S, Amani MH, Mondelli M. 2022. Memorization and optimization in deep neural networks with minimum over-parameterization. 36th Conference on Neural Information Processing Systems. NeurIPS: Neural Information Processing Systems, Advances in Neural Information Processing Systems, vol. 35, 7628–7640.","chicago":"Bombari, Simone, Mohammad Hossein Amani, and Marco Mondelli. “Memorization and Optimization in Deep Neural Networks with Minimum Over-Parameterization.” In <i>36th Conference on Neural Information Processing Systems</i>, 35:7628–40. Neural Information Processing Systems Foundation, 2022."},"page":"7628-7640","external_id":{"arxiv":["2205.10217"]},"main_file_link":[{"url":" https://doi.org/10.48550/arXiv.2205.10217","open_access":"1"}],"language":[{"iso":"eng"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","date_created":"2023-02-10T13:46:37Z","oa_version":"Preprint","intvolume":"        35","quality_controlled":"1","arxiv":1,"type":"conference","date_published":"2022-07-24T00:00:00Z","date_updated":"2026-09-21T13:07:01Z","department":[{"_id":"MaMo"}],"publication_status":"published"}]
