[{"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"},{"grant_number":"101161364","name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf"},{"_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A","name":"Improving estimation and prediction of common complex disease risk","grant_number":"PCEGP3_181181"}],"author":[{"last_name":"Depope","first_name":"Al","full_name":"Depope, Al","id":"0b77531d-dbcd-11ea-9d1d-a8eee0bf3830"}],"year":"2026","file_date_updated":"2026-07-13T14:56:41Z","oa_version":"Published Version","supervisor":[{"first_name":"Matthew Richard","orcid":"0000-0001-8982-8813","last_name":"Robinson","id":"E5D42276-F5DA-11E9-8E24-6303E6697425","full_name":"Robinson, Matthew Richard"},{"first_name":"Marco","orcid":"0000-0002-3242-7020","last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco"}],"day":"11","related_material":{"record":[{"status":"public","relation":"part_of_dissertation","id":"21488"}]},"publisher":"Institute of Science and Technology Austria","file":[{"relation":"main_file","date_created":"2026-07-13T14:52:19Z","content_type":"application/pdf","date_updated":"2026-07-13T14:52:19Z","file_name":"2026_Depope_Al_Thesis.pdf","creator":"adepope","file_size":25109878,"access_level":"open_access","file_id":"22316","checksum":"9ab386790515628d957a194f30a7ccb4"},{"checksum":"8ed8fb63f76a695d5b6fec35343f4b90","file_id":"22317","file_size":1203199939,"access_level":"closed","creator":"adepope","file_name":"2026_Depope_Al_Thesis.zip","date_updated":"2026-07-13T14:56:41Z","content_type":"application/zip","date_created":"2026-07-13T14:56:41Z","relation":"source_file"}],"status":"public","acknowledged_ssus":[{"_id":"ScienComp"}],"has_accepted_license":"1","citation":{"short":"A. Depope, From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics, Institute of Science and Technology Austria, 2026.","mla":"Depope, Al. <i>From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics</i>. Institute of Science and Technology Austria, 2026, doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>.","ista":"Depope A. 2026. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. Institute of Science and Technology Austria.","apa":"Depope, A. (2026). <i>From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>","ama":"Depope A. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. 2026. doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>","ieee":"A. Depope, “From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics,” Institute of Science and Technology Austria, 2026.","chicago":"Depope, Al. “From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics.” Institute of Science and Technology Austria, 2026. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>."},"degree_awarded":"PhD","department":[{"_id":"GradSch"},{"_id":"MaRo"},{"_id":"MaMo"}],"month":"07","user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","article_processing_charge":"No","ddc":["576","610","006"],"_id":"22258","das_tickbox":"1","date_created":"2026-07-10T13:27:20Z","page":"169","type":"dissertation","corr_author":"1","doi_confirm":"1","date_published":"2026-07-11T00:00:00Z","title":"From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics","abstract":[{"text":"Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.\r\n\r\nFirst, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. \r\n\r\nSecond, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.\r\n\r\nCollectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.","lang":"eng"}],"doi":"10.15479/AT-ISTA-22258","publication_identifier":{"issn":["2663-337X"]},"date_updated":"2026-07-28T07:08:15Z","publication_status":"published","alternative_title":["ISTA Thesis"],"oa":1,"language":[{"iso":"eng"}],"OA_place":"publisher","keyword":["Approximate Message Passing","GWAS","Genomics","Proteomics","Survival modeling"],"acknowledgement":"This work was supported in part by the Swiss National Science Foundation through the\r\nEccellenza Grant \"Improving estimation and prediction of common complex disease risk\"\r\n(grant number PCEGP3_181181); the European Research Council through the grant\r\n\"Inference in High Dimensions: Light-speed Algorithms and Information Limits\" (grant\r\nnumber 101161364); and the Fondation Jean-Jacques et Felicia Lopez-Loreta through the\r\nPrix Lopez-Loretta 2019.\r\n"},{"citation":{"apa":"Depope, A., Bajzik, J., Mondelli, M., &#38; Robinson, M. R. (2026). Joint modeling of whole-genome sequencing data for human height via approximate message passing. <i>Cell Genomics</i>. Elsevier. <a href=\"https://doi.org/10.1016/j.xgen.2026.101162\">https://doi.org/10.1016/j.xgen.2026.101162</a>","chicago":"Depope, Al, Jakub Bajzik, Marco Mondelli, and Matthew Richard Robinson. “Joint Modeling of Whole-Genome Sequencing Data for Human Height via Approximate Message Passing.” <i>Cell Genomics</i>. Elsevier, 2026. <a href=\"https://doi.org/10.1016/j.xgen.2026.101162\">https://doi.org/10.1016/j.xgen.2026.101162</a>.","ieee":"A. Depope, J. Bajzik, M. Mondelli, and M. R. Robinson, “Joint modeling of whole-genome sequencing data for human height via approximate message passing,” <i>Cell Genomics</i>, vol. 6, no. 5. Elsevier, 2026.","ama":"Depope A, Bajzik J, Mondelli M, Robinson MR. Joint modeling of whole-genome sequencing data for human height via approximate message passing. <i>Cell Genomics</i>. 2026;6(5). doi:<a href=\"https://doi.org/10.1016/j.xgen.2026.101162\">10.1016/j.xgen.2026.101162</a>","ista":"Depope A, Bajzik J, Mondelli M, Robinson MR. 2026. Joint modeling of whole-genome sequencing data for human height via approximate message passing. Cell Genomics. 6(5), 101162.","mla":"Depope, Al, et al. “Joint Modeling of Whole-Genome Sequencing Data for Human Height via Approximate Message Passing.” <i>Cell Genomics</i>, vol. 6, no. 5, 101162, Elsevier, 2026, doi:<a href=\"https://doi.org/10.1016/j.xgen.2026.101162\">10.1016/j.xgen.2026.101162</a>.","short":"A. Depope, J. Bajzik, M. Mondelli, M.R. Robinson, Cell Genomics 6 (2026)."},"month":"05","department":[{"_id":"MaMo"},{"_id":"MaRo"}],"external_id":{"pmid":["41713425"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","volume":6,"day":"13","article_type":"original","publisher":"Elsevier","scopus_import":"1","file":[{"success":1,"relation":"main_file","date_updated":"2026-07-28T07:06:26Z","date_created":"2026-07-28T07:06:26Z","content_type":"application/pdf","access_level":"open_access","file_size":3736705,"checksum":"6b59686f8d9733add4f23d23f3dd9df0","file_id":"22596","creator":"dernst","file_name":"2026_CellGenomics_Depope.pdf"}],"status":"public","related_material":{"record":[{"id":"22258","relation":"dissertation_contains","status":"public"}],"link":[{"url":"https://ista.ac.at/en/news/big-data-and-human-height/","description":"News on ISTA website","relation":"press_release"}]},"tmp":{"image":"/images/cc_by_nc_nd.png","legal_code_url":"https://creativecommons.org/licenses/by-nc-nd/4.0/legalcode","name":"Creative Commons Attribution-NonCommercial-NoDerivatives 4.0 International (CC BY-NC-ND 4.0)","short":"CC BY-NC-ND (4.0)"},"has_accepted_license":"1","OA_type":"gold","year":"2026","oa_version":"Published Version","article_number":"101162","file_date_updated":"2026-07-28T07:06:26Z","quality_controlled":"1","author":[{"last_name":"Depope","first_name":"Al","full_name":"Depope, Al","id":"0b77531d-dbcd-11ea-9d1d-a8eee0bf3830"},{"id":"b995e25b-8c4b-11ed-a6d8-f71b7bcd6122","full_name":"Bajzik, Jakub","first_name":"Jakub","last_name":"Bajzik"},{"last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"},{"id":"E5D42276-F5DA-11E9-8E24-6303E6697425","full_name":"Robinson, Matthew Richard","first_name":"Matthew Richard","orcid":"0000-0001-8982-8813","last_name":"Robinson"}],"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"},{"_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf","grant_number":"101161364","name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits"},{"name":"Improving estimation and prediction of common complex disease risk","grant_number":"PCEGP3_181181","_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A"}],"oa":1,"dataavailabilitystatement":"This project uses the UK Biobank data under project number 35520. UK Biobank genotypic and phenotypic data are available through a formal request at http://www.ukbiobank.ac.uk. It also uses genotypic and phenotypic data from the All of Us study, which are also available through a formal request at https://www.researchallofus.org/data-tools/data-access/. All summary statistic estimates are released publicly on Dryad: https://doi.org/10.5061/dryad.cz8w9gjjc.\r\n•\r\nThe gVAMP code developed in this work is open source and has been deposited on GitHub, where it is publicly available at https://github.com/medical-genomics-group/gVAMP, and the code used to generate the data in the manuscript are available from Zenodo https://doi.org/10.5281/zenodo.17935521. The URLs of other software used are listed in the key resources table.","publication":"Cell Genomics","publication_status":"published","DOAJ_listed":"1","OA_place":"publisher","language":[{"iso":"eng"}],"acknowledgement":"We thank Malgorzata Borczyk for creating the gene burden scores. We thank Robin Beaumont, Amedeo Roberto Esposito, Gareth Hawkes, Philip Schniter, Matthew Stephens, Pragya Sur, Peter Visscher, Michael Weedon, and Harry Wright for providing valuable suggestions and comments on earlier versions of the work. This project was funded by a Lopez-Loreta Prize to M.M., an SNSF Eccellenza Grant to M.R.R. (PCEGP3-181181), an ERC Starting Grant to M.M. (INF2, project number 101161364), and core funding from ISTA. High-performance computing was supported by the Scientific Service Units (SSU) of ISTA through resources provided by Scientific Computing (SciComp). We would like to acknowledge the participants and investigators of the UK Biobank study. We gratefully acknowledge the All of Us participants for their contributions, without whom this research would not have been possible. We also thank the National Institutes of Health All of Us Research Program for making available the participant data (and/or samples and/or cohort) examined in this study.","pmid":1,"title":"Joint modeling of whole-genome sequencing data for human height via approximate message passing","date_published":"2026-05-13T00:00:00Z","abstract":[{"lang":"eng","text":"Human height is a model for the genetic analysis of complex traits, and recent studies suggest the presence of thousands of common genetic variant associations and hundreds of low-frequency/rare variants. Here, we develop a new algorithmic paradigm based on approximate message passing (genomic vector approximate message passing [gVAMP]) for identifying DNA sequence variants associated with complex traits and common diseases in large-scale whole-genome sequencing (WGS) data. We show that gVAMP accurately localizes associations to variants with the correct frequency and position in the DNA, outperforming existing fine-mapping methods in selecting the appropriate genetic variants within WGS data. We then apply gVAMP to jointly model the relationship of tens of millions of WGS variants with human height in hundreds of thousands of UK Biobank individuals. We identify 59 rare variants and gene burden scores alongside many hundreds of DNA regions containing common variant associations and show that understanding the genetic basis of complex traits will require the joint analysis of hundreds of millions of variables measured on millions of people. The polygenic risk scores obtained from gVAMP have high accuracy (including a prediction accuracy of ∼46% for human height) and outperform current methods for downstream tasks such as mixed linear model association testing across 13 UK Biobank traits. In conclusion, gVAMP offers a scalable foundation for a wider range of analyses in WGS data."}],"license":"https://creativecommons.org/licenses/by-nc-nd/4.0/","publication_identifier":{"eissn":["2666-979X"]},"doi":"10.1016/j.xgen.2026.101162","date_updated":"2026-07-28T07:08:15Z","researchdata_availability":"yes","issue":"5","corr_author":"1","intvolume":"         6","article_processing_charge":"Yes","_id":"21488","das_tickbox":"1","ddc":["000","570"],"date_created":"2026-03-23T15:10:03Z","type":"journal_article","supplementarymaterial":"yes"},{"intvolume":"         8","corr_author":"1","PlanS_conform":"1","issue":"2","researchdata_availability":"no","type":"journal_article","supplementarymaterial":"no","date_created":"2026-06-30T13:03:41Z","page":"411-439","ddc":["000"],"_id":"22228","das_tickbox":"0","arxiv":1,"article_processing_charge":"Yes (in subscription journal)","keyword":["spectral estimator","generalized linear models","mixed regression","high-dimensional asymptotics","random matrix theory","approximate message passing (AMP)"],"acknowledgement":"The first and second authors were partially supported by the 2019 Lopez-Loreta prize.","language":[{"iso":"eng"}],"OA_place":"publisher","publication_status":"published","publication":"SIAM Journal on Mathematics of Data Science","oa":1,"date_updated":"2026-08-12T06:23:07Z","doi":"10.1137/24m1702854","publication_identifier":{"eissn":["2577-0187"]},"abstract":[{"lang":"eng","text":"In a mixed generalized linear model, the goal is to learn multiple signals from unlabeled observations: each sample comes from exactly one signal, but it is not known which one. We consider the prototypical problem of estimating two statistically independent signals in a mixed generalized linear model with Gaussian covariates. Spectral methods are a popular class of estimators which output the top two eigenvectors of a suitable data-dependent matrix. However, despite the wide applicability, their design is still obtained via heuristic considerations, and the number of samples 𝑛 needed to guarantee recovery is superlinear in the signal dimension 𝑑. In this paper, we develop exact asymptotics on spectral methods in the challenging proportional regime in which 𝑛,𝑑 grow large and their ratio converges to a finite constant. This allows us optimize the design of the spectral method, and combine it with a simple linear estimator, to minimize the estimation error. Our characterization exploits a mix of tools from random matrices, free probability, and the theory of approximate message passing algorithms. Numerical simulations for mixed linear regression and phase retrieval demonstrate the advantage enabled by our analysis over existing designs of spectral methods."}],"license":"https://creativecommons.org/licenses/by/4.0/","date_published":"2026-06-01T00:00:00Z","title":"Precise asymptotics for spectral methods in mixed generalized linear models","quality_controlled":"1","file_date_updated":"2026-07-01T06:22:15Z","oa_version":"Published Version","year":"2026","OA_type":"hybrid","author":[{"first_name":"Yihan","last_name":"Zhang","full_name":"Zhang, Yihan"},{"last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"},{"full_name":"Venkataramanan, Ramji","first_name":"Ramji","last_name":"Venkataramanan"}],"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"volume":8,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","external_id":{"arxiv":["2211.11368"]},"department":[{"_id":"MaMo"}],"month":"06","citation":{"ista":"Zhang Y, Mondelli M, Venkataramanan R. 2026. Precise asymptotics for spectral methods in mixed generalized linear models. SIAM Journal on Mathematics of Data Science. 8(2), 411–439.","apa":"Zhang, Y., Mondelli, M., &#38; Venkataramanan, R. (2026). Precise asymptotics for spectral methods in mixed generalized linear models. <i>SIAM Journal on Mathematics of Data Science</i>. SIAM. <a href=\"https://doi.org/10.1137/24m1702854\">https://doi.org/10.1137/24m1702854</a>","chicago":"Zhang, Yihan, Marco Mondelli, and Ramji Venkataramanan. “Precise Asymptotics for Spectral Methods in Mixed Generalized Linear Models.” <i>SIAM Journal on Mathematics of Data Science</i>. SIAM, 2026. <a href=\"https://doi.org/10.1137/24m1702854\">https://doi.org/10.1137/24m1702854</a>.","ama":"Zhang Y, Mondelli M, Venkataramanan R. Precise asymptotics for spectral methods in mixed generalized linear models. <i>SIAM Journal on Mathematics of Data Science</i>. 2026;8(2):411-439. doi:<a href=\"https://doi.org/10.1137/24m1702854\">10.1137/24m1702854</a>","ieee":"Y. Zhang, M. Mondelli, and R. Venkataramanan, “Precise asymptotics for spectral methods in mixed generalized linear models,” <i>SIAM Journal on Mathematics of Data Science</i>, vol. 8, no. 2. SIAM, pp. 411–439, 2026.","short":"Y. Zhang, M. Mondelli, R. Venkataramanan, SIAM Journal on Mathematics of Data Science 8 (2026) 411–439.","mla":"Zhang, Yihan, et al. “Precise Asymptotics for Spectral Methods in Mixed Generalized Linear Models.” <i>SIAM Journal on Mathematics of Data Science</i>, vol. 8, no. 2, SIAM, 2026, pp. 411–39, doi:<a href=\"https://doi.org/10.1137/24m1702854\">10.1137/24m1702854</a>."},"mathsc":["62E20","62J05","62J12"],"has_accepted_license":"1","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"scopus_import":"1","status":"public","file":[{"success":1,"relation":"main_file","date_updated":"2026-07-01T06:22:15Z","date_created":"2026-07-01T06:22:15Z","content_type":"application/pdf","access_level":"open_access","file_size":1210346,"checksum":"5cfd350dc64d1476063e959316dbff65","file_id":"22230","file_name":"2026_SIAMJourmathDataScience_Zhang.pdf","creator":"dernst"}],"publisher":"SIAM","article_type":"original","day":"01"},{"intvolume":"         7","arxiv":1,"article_processing_charge":"Yes","ddc":["530"],"_id":"18986","date_created":"2025-02-02T23:01:54Z","type":"journal_article","DOAJ_listed":"1","publication":"Physical Review Research","publication_status":"published","oa":1,"language":[{"iso":"eng"}],"OA_place":"publisher","acknowledgement":"J.B., F.C., and Y.X. were funded by the European Union (ERC, CHORAL, Project No. 101039794). Views and opinions expressed are however those of the authors only and do not necessarily reflect those of the European Union or the European Research Council. Neither the European Union nor the granting authority can be held responsible for them. M.M. was supported by the 2019 Lopez-Loreta Prize. J.B. acknowledges discussions with TianQi Hou at the initial stage of the project, as well as with Antoine Bodin.","date_published":"2025-01-22T00:00:00Z","APC_amount":"3272,21 EUR","title":"Information limits and Thouless-Anderson-Palmer equations for spiked matrix models with structured noise","abstract":[{"lang":"eng","text":"We consider a prototypical problem of Bayesian inference for a structured spiked model: a low-rank signal is corrupted by additive noise. While both information-theoretic and algorithmic limits are well understood when the noise is a Gaussian Wigner matrix, the more realistic case of structured noise still remains challenging. To capture the structure while maintaining mathematical tractability, a line of work has focused on rotationally invariant noise. However, existing studies either provide suboptimal algorithms or are limited to a special class of noise ensembles. In this paper, using tools from statistical physics (replica method) and random matrix theory (generalized spherical integrals) we establish the characterization of the information-theoretic limits for a noise matrix drawn from a general trace ensemble. Remarkably, our analysis unveils the asymptotic equivalence between the rotationally invariant model and a surrogate Gaussian one. Finally, we show how to saturate the predicted statistical limits using an efficient algorithm inspired by the theory of adaptive Thouless-Anderson-Palmer (TAP) equations."}],"doi":"10.1103/PhysRevResearch.7.013081","publication_identifier":{"issn":["2643-1564"]},"date_updated":"2026-05-06T12:57:36Z","year":"2025","OA_type":"gold","file_date_updated":"2025-02-03T08:27:59Z","oa_version":"Published Version","article_number":"013081","quality_controlled":"1","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"author":[{"last_name":"Barbier","first_name":"Jean","full_name":"Barbier, Jean"},{"first_name":"Francesco","last_name":"Camilli","full_name":"Camilli, Francesco"},{"full_name":"Xu, Yizhou","first_name":"Yizhou","last_name":"Xu"},{"orcid":"0000-0002-3242-7020","first_name":"Marco","last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco"}],"citation":{"ista":"Barbier J, Camilli F, Xu Y, Mondelli M. 2025. Information limits and Thouless-Anderson-Palmer equations for spiked matrix models with structured noise. Physical Review Research. 7, 013081.","apa":"Barbier, J., Camilli, F., Xu, Y., &#38; Mondelli, M. (2025). Information limits and Thouless-Anderson-Palmer equations for spiked matrix models with structured noise. <i>Physical Review Research</i>. American Physical Society. <a href=\"https://doi.org/10.1103/PhysRevResearch.7.013081\">https://doi.org/10.1103/PhysRevResearch.7.013081</a>","chicago":"Barbier, Jean, Francesco Camilli, Yizhou Xu, and Marco Mondelli. “Information Limits and Thouless-Anderson-Palmer Equations for Spiked Matrix Models with Structured Noise.” <i>Physical Review Research</i>. American Physical Society, 2025. <a href=\"https://doi.org/10.1103/PhysRevResearch.7.013081\">https://doi.org/10.1103/PhysRevResearch.7.013081</a>.","ama":"Barbier J, Camilli F, Xu Y, Mondelli M. Information limits and Thouless-Anderson-Palmer equations for spiked matrix models with structured noise. <i>Physical Review Research</i>. 2025;7. doi:<a href=\"https://doi.org/10.1103/PhysRevResearch.7.013081\">10.1103/PhysRevResearch.7.013081</a>","ieee":"J. Barbier, F. Camilli, Y. Xu, and M. Mondelli, “Information limits and Thouless-Anderson-Palmer equations for spiked matrix models with structured noise,” <i>Physical Review Research</i>, vol. 7. American Physical Society, 2025.","short":"J. Barbier, F. Camilli, Y. Xu, M. Mondelli, Physical Review Research 7 (2025).","mla":"Barbier, Jean, et al. “Information Limits and Thouless-Anderson-Palmer Equations for Spiked Matrix Models with Structured Noise.” <i>Physical Review Research</i>, vol. 7, 013081, American Physical Society, 2025, doi:<a href=\"https://doi.org/10.1103/PhysRevResearch.7.013081\">10.1103/PhysRevResearch.7.013081</a>."},"department":[{"_id":"MaMo"}],"month":"01","external_id":{"arxiv":["2405.20993"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","volume":7,"day":"22","article_type":"original","related_material":{"link":[{"url":"https://github.com/xu-yz19/spiked-matrix-models-with-structured-noise","relation":"software"}]},"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"file":[{"relation":"main_file","success":1,"content_type":"application/pdf","date_created":"2025-02-03T08:27:59Z","date_updated":"2025-02-03T08:27:59Z","file_name":"2025_PhysReviewResearch_Barbier.pdf","creator":"dernst","checksum":"52c5f72d80ffc928542469114fcdb62b","file_id":"18988","access_level":"open_access","file_size":702543}],"status":"public","publisher":"American Physical Society","scopus_import":"1","has_accepted_license":"1"},{"citation":{"apa":"Bombari, S., &#38; Mondelli, M. (2025). Privacy for free in the overparameterized regime. <i>Proceedings of the National Academy of Sciences</i>. National Academy of Sciences. <a href=\"https://doi.org/10.1073/pnas.2423072122\">https://doi.org/10.1073/pnas.2423072122</a>","ama":"Bombari S, Mondelli M. Privacy for free in the overparameterized regime. <i>Proceedings of the National Academy of Sciences</i>. 2025;122(15). doi:<a href=\"https://doi.org/10.1073/pnas.2423072122\">10.1073/pnas.2423072122</a>","chicago":"Bombari, Simone, and Marco Mondelli. “Privacy for Free in the Overparameterized Regime.” <i>Proceedings of the National Academy of Sciences</i>. National Academy of Sciences, 2025. <a href=\"https://doi.org/10.1073/pnas.2423072122\">https://doi.org/10.1073/pnas.2423072122</a>.","ieee":"S. Bombari and M. Mondelli, “Privacy for free in the overparameterized regime,” <i>Proceedings of the National Academy of Sciences</i>, vol. 122, no. 15. National Academy of Sciences, 2025.","ista":"Bombari S, Mondelli M. 2025. Privacy for free in the overparameterized regime. Proceedings of the National Academy of Sciences. 122(15), e2423072122.","mla":"Bombari, Simone, and Marco Mondelli. “Privacy for Free in the Overparameterized Regime.” <i>Proceedings of the National Academy of Sciences</i>, vol. 122, no. 15, e2423072122, National Academy of Sciences, 2025, doi:<a href=\"https://doi.org/10.1073/pnas.2423072122\">10.1073/pnas.2423072122</a>.","short":"S. Bombari, M. Mondelli, Proceedings of the National Academy of Sciences 122 (2025)."},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","volume":122,"external_id":{"arxiv":["2410.14787"],"isi":["001471214000001"],"pmid":["40215275"]},"department":[{"_id":"MaMo"}],"month":"04","article_type":"original","day":"15","has_accepted_license":"1","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"scopus_import":"1","publisher":"National Academy of Sciences","file":[{"relation":"main_file","success":1,"creator":"dernst","file_name":"2025_PNAS_Bombari.pdf","checksum":"1ac6f78e368d35a0cafb4d2d9bd63443","file_id":"19648","access_level":"open_access","file_size":2328320,"content_type":"application/pdf","date_created":"2025-05-05T07:27:54Z","date_updated":"2025-05-05T07:27:54Z"}],"status":"public","file_date_updated":"2025-05-05T07:27:54Z","oa_version":"Published Version","article_number":"e2423072122","year":"2025","OA_type":"hybrid","quality_controlled":"1","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"},{"name":"Trustworthy Deep Learning Theory: Private Over-Parameterized Models and Robust LLMs","_id":"92099302-16d5-11f0-9cad-f9a785f54fbd"}],"author":[{"full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","last_name":"Bombari","first_name":"Simone"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"isi":1,"language":[{"iso":"eng"}],"OA_place":"publisher","publication":"Proceedings of the National Academy of Sciences","publication_status":"published","oa":1,"pmid":1,"acknowledgement":"This research was funded in whole, or in part, by the Austrian Science Fund (FWF) Grant number COE 12. For the purpose of open access, the author has applied a CC BY public copyright license to any Author Accepted Manuscript version arising from this submission. The authors were also supported by the 2019 Lopez-Loreta prize, and Simone Bombari was supported by a Google PhD fellowship. We thank Diyuan Wu, Edwige Cyffers, Francesco Pedrotti, Inbar Seroussi, Nikita P. Kalinin, Pietro Pelliconi, Roodabeh Safavi, Yizhe Zhu, and Zhichao Wang for helpful discussions.","abstract":[{"lang":"eng","text":"Differentially private gradient descent (DP-GD) is a popular algorithm to train deep learning models with provable guarantees on the privacy of the training data. In the last decade, the problem of understanding its performance cost with respect to standard GD has received remarkable attention from the research community, which formally derived upper bounds on the excess population risk  RP  in different learning settings. However, existing bounds typically degrade with over-parameterization, i.e., as the number of parameters  p  gets larger than the number of training samples  n  -- a regime which is ubiquitous in current deep-learning practice. As a result, the lack of theoretical insights leaves practitioners without clear guidance, leading some to reduce the effective number of trainable parameters to improve performance, while others use larger models to achieve better results through scale. In this work, we show that in the popular random features model with quadratic loss, for any sufficiently large  p , privacy can be obtained for free, i.e.,  |RP|=o(1) , not only when the privacy parameter  ε  has constant order, but also in the strongly private setting  ε=o(1) . This challenges the common wisdom that over-parameterization inherently hinders performance in private learning."}],"date_published":"2025-04-15T00:00:00Z","APC_amount":"2754,32 EUR","title":"Privacy for free in the overparameterized regime","date_updated":"2026-05-20T08:23:19Z","doi":"10.1073/pnas.2423072122","publication_identifier":{"eissn":["1091-6490"],"issn":["0027-8424"]},"issue":"15","intvolume":"       122","corr_author":"1","ddc":["000"],"_id":"19627","arxiv":1,"article_processing_charge":"Yes (in subscription journal)","type":"journal_article","date_created":"2025-04-27T22:02:13Z"},{"type":"journal_article","page":"193-304","date_created":"2025-12-07T23:02:02Z","ddc":["000"],"_id":"20734","article_processing_charge":"No","intvolume":"         8","PlanS_conform":"1","corr_author":"1","issue":"3-4","date_updated":"2025-12-09T13:53:31Z","publication_identifier":{"eissn":["2520-2324"],"issn":["2520-2316"]},"doi":"10.4171/MSL/52","abstract":[{"text":"We consider the problem of parameter estimation in a high-dimensional generalized linear model. Spectral methods obtained via the principal eigenvector of a suitable data-dependent matrix provide a simple yet surprisingly effective solution. However, despite their wide use, a rigorous performance characterization, as well as a principled way to preprocess the data, are available only for unstructured (i.i.d. Gaussian and Haar orthogonal) designs. In contrast, real-world data matrices are highly structured and exhibit non-trivial correlations. To address the problem, we consider correlated Gaussian designs capturing the anisotropic nature of the features via a covariance matrix Σ. Our main result is a precise asymptotic characterization of the performance of spectral estimators. This allows us to identify the optimal preprocessing that minimizes the number of samples needed for parameter estimation. Surprisingly, such preprocessing is universal across a broad set of designs, which partly addresses a conjecture on optimal spectral estimators for rotationally invariant models. Our principled approach vastly improves upon previous heuristic methods, including for designs common in computational imaging and genetics. The proposed methodology, based on approximate message passing, is broadly applicable and opens the way to the precise characterization of spiked matrices and of the corresponding spectral methods in a variety of settings.","lang":"eng"}],"title":"Spectral estimators for structured generalized linear models via approximate message passing","date_published":"2025-09-02T00:00:00Z","acknowledgement":"This work was done when Y. Z. and H. C. J. were at the Institute of Science and Technology Austria. Y. Z. thanks Hugo Latourelle-Vigeant for bringing [53] to the authors’ attention.\r\nY. Z. and M. M. are partially supported by the 2019 Lopez-Loreta Prize and by the Interdisciplinary Projects Committee (IPC) at ISTA. H. C. J. is supported by the ERC Advanced Grant “RMTBeyond” No. 101020331.","OA_place":"publisher","language":[{"iso":"eng"}],"oa":1,"publication":"Mathematical Statistics and Learning","publication_status":"published","project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"author":[{"last_name":"Zhang","first_name":"Yihan","orcid":"0000-0002-6465-6258","full_name":"Zhang, Yihan","id":"2ce5da42-b2ea-11eb-bba5-9f264e9d002c"},{"full_name":"Ji, Hong Chang","last_name":"Ji","first_name":"Hong Chang"},{"first_name":"Ramji","last_name":"Venkataramanan","full_name":"Venkataramanan, Ramji"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"quality_controlled":"1","oa_version":"Published Version","file_date_updated":"2025-12-09T13:50:03Z","OA_type":"diamond","year":"2025","has_accepted_license":"1","scopus_import":"1","status":"public","publisher":"EMS Press","file":[{"content_type":"application/pdf","date_created":"2025-12-09T13:50:03Z","date_updated":"2025-12-09T13:50:03Z","file_name":"2025_MathStatLearning_Zhang.pdf","creator":"dernst","file_id":"20752","checksum":"55a1bd9c1b6b0198c42504fb94f4ad4c","access_level":"open_access","file_size":1379626,"relation":"main_file","success":1}],"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"day":"02","article_type":"original","volume":8,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","month":"09","department":[{"_id":"MaMo"}],"citation":{"short":"Y. Zhang, H.C. Ji, R. Venkataramanan, M. Mondelli, Mathematical Statistics and Learning 8 (2025) 193–304.","mla":"Zhang, Yihan, et al. “Spectral Estimators for Structured Generalized Linear Models via Approximate Message Passing.” <i>Mathematical Statistics and Learning</i>, vol. 8, no. 3–4, EMS Press, 2025, pp. 193–304, doi:<a href=\"https://doi.org/10.4171/MSL/52\">10.4171/MSL/52</a>.","ista":"Zhang Y, Ji HC, Venkataramanan R, Mondelli M. 2025. Spectral estimators for structured generalized linear models via approximate message passing. Mathematical Statistics and Learning. 8(3–4), 193–304.","apa":"Zhang, Y., Ji, H. C., Venkataramanan, R., &#38; Mondelli, M. (2025). Spectral estimators for structured generalized linear models via approximate message passing. <i>Mathematical Statistics and Learning</i>. EMS Press. <a href=\"https://doi.org/10.4171/MSL/52\">https://doi.org/10.4171/MSL/52</a>","chicago":"Zhang, Yihan, Hong Chang Ji, Ramji Venkataramanan, and Marco Mondelli. “Spectral Estimators for Structured Generalized Linear Models via Approximate Message Passing.” <i>Mathematical Statistics and Learning</i>. EMS Press, 2025. <a href=\"https://doi.org/10.4171/MSL/52\">https://doi.org/10.4171/MSL/52</a>.","ama":"Zhang Y, Ji HC, Venkataramanan R, Mondelli M. Spectral estimators for structured generalized linear models via approximate message passing. <i>Mathematical Statistics and Learning</i>. 2025;8(3-4):193-304. doi:<a href=\"https://doi.org/10.4171/MSL/52\">10.4171/MSL/52</a>","ieee":"Y. Zhang, H. C. Ji, R. Venkataramanan, and M. Mondelli, “Spectral estimators for structured generalized linear models via approximate message passing,” <i>Mathematical Statistics and Learning</i>, vol. 8, no. 3–4. EMS Press, pp. 193–304, 2025."}},{"article_type":"original","day":"01","status":"public","scopus_import":"1","publisher":"IEEE","related_material":{"record":[{"id":"14922","relation":"earlier_version","status":"public"}]},"citation":{"ista":"Esposito AR, Mondelli M. 2024. Concentration without independence via information measures. IEEE Transactions on Information Theory. 70(6), 3823–3839.","apa":"Esposito, A. R., &#38; Mondelli, M. (2024). Concentration without independence via information measures. <i>IEEE Transactions on Information Theory</i>. IEEE. <a href=\"https://doi.org/10.1109/TIT.2024.3367767\">https://doi.org/10.1109/TIT.2024.3367767</a>","chicago":"Esposito, Amedeo Roberto, and Marco Mondelli. “Concentration without Independence via Information Measures.” <i>IEEE Transactions on Information Theory</i>. IEEE, 2024. <a href=\"https://doi.org/10.1109/TIT.2024.3367767\">https://doi.org/10.1109/TIT.2024.3367767</a>.","ama":"Esposito AR, Mondelli M. Concentration without independence via information measures. <i>IEEE Transactions on Information Theory</i>. 2024;70(6):3823-3839. doi:<a href=\"https://doi.org/10.1109/TIT.2024.3367767\">10.1109/TIT.2024.3367767</a>","ieee":"A. R. Esposito and M. Mondelli, “Concentration without independence via information measures,” <i>IEEE Transactions on Information Theory</i>, vol. 70, no. 6. IEEE, pp. 3823–3839, 2024.","short":"A.R. Esposito, M. Mondelli, IEEE Transactions on Information Theory 70 (2024) 3823–3839.","mla":"Esposito, Amedeo Roberto, and Marco Mondelli. “Concentration without Independence via Information Measures.” <i>IEEE Transactions on Information Theory</i>, vol. 70, no. 6, IEEE, 2024, pp. 3823–39, doi:<a href=\"https://doi.org/10.1109/TIT.2024.3367767\">10.1109/TIT.2024.3367767</a>."},"month":"06","department":[{"_id":"MaMo"}],"external_id":{"isi":["001230181100001"],"arxiv":["2303.07245"]},"user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","volume":70,"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"author":[{"id":"9583e921-e1ad-11ec-9862-cef099626dc9","full_name":"Esposito, Amedeo Roberto","first_name":"Amedeo Roberto","last_name":"Esposito"},{"last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"}],"isi":1,"year":"2024","oa_version":"Preprint","quality_controlled":"1","title":"Concentration without independence via information measures","date_published":"2024-06-01T00:00:00Z","abstract":[{"lang":"eng","text":"We propose a novel approach to concentration for non-independent random variables. The main idea is to “pretend” that the random variables are independent and pay a multiplicative price measuring how far they are from actually being independent. This price is encapsulated in the Hellinger integral between the joint and the product of the marginals, which is then upper bounded leveraging tensorisation properties. Our bounds represent a natural generalisation of concentration inequalities in the presence of dependence: we recover exactly the classical bounds (McDiarmid’s inequality) when the random variables are independent. Furthermore, in a “large deviations” regime, we obtain the same decay in the probability as for the independent case, even when the random variables display non-trivial dependencies. To show this, we consider a number of applications of interest. First, we provide a bound for Markov chains with finite state space. Then, we consider the Simple Symmetric Random Walk, which is a non-contracting Markov chain, and a non-Markovian setting in which the stochastic process depends on its entire past. To conclude, we propose an application to Markov Chain Monte Carlo methods, where our approach leads to an improved lower bound on the minimum burn-in period required to reach a certain accuracy. In all of these settings, we provide a regime of parameters in which our bound fares better than what the state of the art can provide."}],"publication_identifier":{"issn":["0018-9448"],"eissn":["1557-9654"]},"doi":"10.1109/TIT.2024.3367767","date_updated":"2025-09-04T13:06:53Z","oa":1,"publication":"IEEE Transactions on Information Theory","publication_status":"published","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2303.07245","open_access":"1"}],"article_processing_charge":"No","arxiv":1,"_id":"15172","page":"3823-3839","date_created":"2024-03-24T23:01:00Z","type":"journal_article","issue":"6","corr_author":"1","intvolume":"        70"},{"intvolume":"        37","corr_author":"1","_id":"18890","article_processing_charge":"No","arxiv":1,"type":"conference","date_created":"2025-01-27T11:11:40Z","OA_place":"repository","language":[{"iso":"eng"}],"oa":1,"alternative_title":["Advances in Neural Information Processing Systems"],"publication":"38th Annual Conference on Neural Information Processing Systems","publication_status":"published","main_file_link":[{"open_access":"1","url":"https://openreview.net/forum?id=lJ1jdl2K9k"}],"acknowledgement":"We acknowledge support from the National Science Foundation (NSF) and the Simons Foundation for the Collaboration on the Theoretical Foundations of Deep Learning through awards DMS-2031883 and #814639 as well as the TILOS institute (NSF CCF-2112665). This work used the programs (1) XSEDE (Extreme science and engineering discovery environment) which is supported by NSF grant numbers ACI-1548562, and (2) ACCESS (Advanced cyberinfrastructure coordination ecosystem: services & support) which is supported by NSF grants numbers #2138259, #2138286, #2138307, #2137603, and #2138296. Specifically, we used the resources from SDSC Expanse GPU compute nodes, and NCSA Delta system, via allocations TG-CIS220009. Marco Mondelli is supported by the 2019 Lopez-Loreta prize. We also acknowledge useful feedback from anonymous reviewers. ","abstract":[{"text":"Deep Neural Collapse (DNC) refers to the surprisingly rigid structure of the data representations in the final layers of Deep Neural Networks (DNNs). Though the phenomenon has been measured in a variety of settings, its emergence is typically explained via data-agnostic approaches, such as the unconstrained features model. In this work, we introduce a data-dependent setting where DNC forms due to feature learning through the average gradient outer product (AGOP). The AGOP is defined with respect to a learned predictor and is equal to the uncentered covariance matrix of its input-output gradients averaged over the training dataset. The Deep Recursive Feature Machine (Deep RFM) is a method that constructs a neural network by iteratively mapping the data with the AGOP and applying an untrained random feature map. We demonstrate empirically that DNC occurs in Deep RFM across standard settings as a consequence of the projection with the AGOP matrix computed at each layer. Further, we theoretically explain DNC in Deep RFM in an asymptotic setting and as a result of kernel learning. We then provide evidence that this mechanism holds for neural networks more generally. In particular, we show that the right singular vectors and values of the weights can be responsible for the majority of within-class variability collapse for DNNs trained in the feature learning regime. As observed in recent work, this singular structure is highly correlated with that of the AGOP.","lang":"eng"}],"title":"Average gradient outer product as a mechanism for deep neural collapse","date_published":"2024-12-01T00:00:00Z","date_updated":"2025-05-14T11:29:45Z","publication_identifier":{"eissn":["1049-5258"]},"oa_version":"Preprint","OA_type":"green","year":"2024","quality_controlled":"1","project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"author":[{"full_name":"Beaglehole, Daniel","first_name":"Daniel","last_name":"Beaglehole"},{"full_name":"Súkeník, Peter","id":"d64d6a8d-eb8e-11eb-b029-96fd216dec3c","last_name":"Súkeník","first_name":"Peter"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco"},{"last_name":"Belkin","first_name":"Mikhail","full_name":"Belkin, Mikhail"}],"citation":{"short":"D. Beaglehole, P. Súkeník, M. Mondelli, M. Belkin, in:, 38th Annual Conference on Neural Information Processing Systems, Neural Information Processing Systems Foundation, 2024.","mla":"Beaglehole, Daniel, et al. “Average Gradient Outer Product as a Mechanism for Deep Neural Collapse.” <i>38th Annual Conference on Neural Information Processing Systems</i>, vol. 37, Neural Information Processing Systems Foundation, 2024.","ista":"Beaglehole D, Súkeník P, Mondelli M, Belkin M. 2024. Average gradient outer product as a mechanism for deep neural collapse. 38th Annual Conference on Neural Information Processing Systems. NeurIPS: Neural Information Processing Systems, Advances in Neural Information Processing Systems, vol. 37.","apa":"Beaglehole, D., Súkeník, P., Mondelli, M., &#38; Belkin, M. (2024). Average gradient outer product as a mechanism for deep neural collapse. In <i>38th Annual Conference on Neural Information Processing Systems</i> (Vol. 37). Vancouver, Canada: Neural Information Processing Systems Foundation.","ama":"Beaglehole D, Súkeník P, Mondelli M, Belkin M. Average gradient outer product as a mechanism for deep neural collapse. In: <i>38th Annual Conference on Neural Information Processing Systems</i>. Vol 37. Neural Information Processing Systems Foundation; 2024.","ieee":"D. Beaglehole, P. Súkeník, M. Mondelli, and M. Belkin, “Average gradient outer product as a mechanism for deep neural collapse,” in <i>38th Annual Conference on Neural Information Processing Systems</i>, Vancouver, Canada, 2024, vol. 37.","chicago":"Beaglehole, Daniel, Peter Súkeník, Marco Mondelli, and Mikhail Belkin. “Average Gradient Outer Product as a Mechanism for Deep Neural Collapse.” In <i>38th Annual Conference on Neural Information Processing Systems</i>, Vol. 37. Neural Information Processing Systems Foundation, 2024."},"conference":{"name":"NeurIPS: Neural Information Processing Systems","location":"Vancouver, Canada","start_date":"2024-12-16","end_date":"2024-12-16"},"external_id":{"arxiv":["2402.13728"]},"volume":37,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","month":"12","department":[{"_id":"GradSch"},{"_id":"MaMo"}],"day":"01","status":"public","publisher":"Neural Information Processing Systems Foundation","scopus_import":"1"},{"intvolume":"        37","corr_author":"1","_id":"18891","ddc":["000"],"article_processing_charge":"No","arxiv":1,"type":"conference","date_created":"2025-01-27T11:15:18Z","OA_place":"publisher","language":[{"iso":"eng"}],"oa":1,"alternative_title":["Advances in Neural Information Processing Systems"],"publication_status":"published","publication":"38th Annual Conference on Neural Information Processing Systems","acknowledgement":"Marco Mondelli is partially supported by the 2019 Lopez-Loreta prize. This research was supported by the Scientific Service Units (SSU) of ISTA through resources provided by Scientific Computing (SciComp).","abstract":[{"lang":"eng","text":"Deep neural networks (DNNs) exhibit a surprising structure in their final layer\r\nknown as neural collapse (NC), and a growing body of works has currently investigated the propagation of neural collapse to earlier layers of DNNs – a phenomenon\r\ncalled deep neural collapse (DNC). However, existing theoretical results are restricted to special cases: linear models, only two layers or binary classification.\r\nIn contrast, we focus on non-linear models of arbitrary depth in multi-class classification and reveal a surprising qualitative shift. As soon as we go beyond two\r\nlayers or two classes, DNC stops being optimal for the deep unconstrained features\r\nmodel (DUFM) – the standard theoretical framework for the analysis of collapse.\r\nThe main culprit is a low-rank bias of multi-layer regularization schemes: this bias\r\nleads to optimal solutions of even lower rank than the neural collapse. We support\r\nour theoretical findings with experiments on both DUFM and real data, which show\r\nthe emergence of the low-rank structure in the solution found by gradient descent."}],"title":"Neural collapse versus low-rank bias: Is deep neural collapse really optimal?","date_published":"2024-12-01T00:00:00Z","date_updated":"2025-06-04T07:19:21Z","oa_version":"Published Version","file_date_updated":"2025-02-04T08:11:25Z","OA_type":"gold","year":"2024","quality_controlled":"1","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"author":[{"full_name":"Súkeník, Peter","id":"d64d6a8d-eb8e-11eb-b029-96fd216dec3c","last_name":"Súkeník","first_name":"Peter"},{"full_name":"Lampert, Christoph","id":"40C20FD2-F248-11E8-B48F-1D18A9856A87","last_name":"Lampert","first_name":"Christoph","orcid":"0000-0001-8622-7887"},{"first_name":"Marco","orcid":"0000-0002-3242-7020","last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco"}],"citation":{"apa":"Súkeník, P., Lampert, C., &#38; Mondelli, M. (2024). Neural collapse versus low-rank bias: Is deep neural collapse really optimal? In <i>38th Annual Conference on Neural Information Processing Systems</i> (Vol. 37). Vancouver, Canada: Neural Information Processing Systems Foundation.","ieee":"P. Súkeník, C. Lampert, and M. Mondelli, “Neural collapse versus low-rank bias: Is deep neural collapse really optimal?,” in <i>38th Annual Conference on Neural Information Processing Systems</i>, Vancouver, Canada, 2024, vol. 37.","ama":"Súkeník P, Lampert C, Mondelli M. Neural collapse versus low-rank bias: Is deep neural collapse really optimal? In: <i>38th Annual Conference on Neural Information Processing Systems</i>. Vol 37. Neural Information Processing Systems Foundation; 2024.","chicago":"Súkeník, Peter, Christoph Lampert, and Marco Mondelli. “Neural Collapse versus Low-Rank Bias: Is Deep Neural Collapse Really Optimal?” In <i>38th Annual Conference on Neural Information Processing Systems</i>, Vol. 37. Neural Information Processing Systems Foundation, 2024.","ista":"Súkeník P, Lampert C, Mondelli M. 2024. Neural collapse versus low-rank bias: Is deep neural collapse really optimal? 38th Annual Conference on Neural Information Processing Systems. NeurIPS: Neural Information Processing Systems, Advances in Neural Information Processing Systems, vol. 37.","mla":"Súkeník, Peter, et al. “Neural Collapse versus Low-Rank Bias: Is Deep Neural Collapse Really Optimal?” <i>38th Annual Conference on Neural Information Processing Systems</i>, vol. 37, Neural Information Processing Systems Foundation, 2024.","short":"P. Súkeník, C. Lampert, M. Mondelli, in:, 38th Annual Conference on Neural Information Processing Systems, Neural Information Processing Systems Foundation, 2024."},"conference":{"end_date":"2024-12-16","start_date":"2024-12-16","location":"Vancouver, Canada","name":"NeurIPS: Neural Information Processing Systems"},"external_id":{"arxiv":["2405.14468"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","volume":37,"month":"12","department":[{"_id":"GradSch"},{"_id":"MaMo"},{"_id":"ChLa"}],"day":"01","acknowledged_ssus":[{"_id":"ScienComp"}],"has_accepted_license":"1","publisher":"Neural Information Processing Systems Foundation","file":[{"success":1,"relation":"main_file","date_created":"2025-02-04T08:11:25Z","content_type":"application/pdf","date_updated":"2025-02-04T08:11:25Z","creator":"dernst","file_name":"2024_NeurIPS_Sukenik.pdf","file_size":1784118,"access_level":"open_access","checksum":"b7b79f1ea3ac1e9e11b3d91faaeb0780","file_id":"18989"}],"status":"public","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"}},{"day":"01","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"related_material":{"record":[{"relation":"earlier_version","id":"17350","status":"public"}]},"scopus_import":"1","file":[{"success":1,"relation":"main_file","file_size":780315,"access_level":"open_access","file_id":"18898","checksum":"76a1fd5afd8ee6f7ae0e5912d7dbf6b4","creator":"dernst","file_name":"2024_TMLR_Pedrotti.pdf","date_updated":"2025-01-27T12:19:44Z","date_created":"2025-01-27T12:19:44Z","content_type":"application/pdf"}],"status":"public","has_accepted_license":"1","citation":{"ista":"Pedrotti F, Maas J, Mondelli M. 2024. Improved convergence of score-based diffusion models via prediction-correction. Transactions on Machine Learning Research. , TMLR, .","apa":"Pedrotti, F., Maas, J., &#38; Mondelli, M. (2024). Improved convergence of score-based diffusion models via prediction-correction. In <i>Transactions on Machine Learning Research</i>.","ama":"Pedrotti F, Maas J, Mondelli M. Improved convergence of score-based diffusion models via prediction-correction. In: <i>Transactions on Machine Learning Research</i>. ; 2024.","chicago":"Pedrotti, Francesco, Jan Maas, and Marco Mondelli. “Improved Convergence of Score-Based Diffusion Models via Prediction-Correction.” In <i>Transactions on Machine Learning Research</i>, 2024.","ieee":"F. Pedrotti, J. Maas, and M. Mondelli, “Improved convergence of score-based diffusion models via prediction-correction,” in <i>Transactions on Machine Learning Research</i>, 2024.","short":"F. Pedrotti, J. Maas, M. Mondelli, in:, Transactions on Machine Learning Research, 2024.","mla":"Pedrotti, Francesco, et al. “Improved Convergence of Score-Based Diffusion Models via Prediction-Correction.” <i>Transactions on Machine Learning Research</i>, 2024."},"department":[{"_id":"JaMa"},{"_id":"MaMo"}],"month":"06","external_id":{"arxiv":["2305.14164"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","author":[{"full_name":"Pedrotti, Francesco","id":"d3ac8ac6-dc8d-11ea-abe3-e2a9628c4c3c","last_name":"Pedrotti","first_name":"Francesco"},{"full_name":"Maas, Jan","id":"4C5696CE-F248-11E8-B48F-1D18A9856A87","last_name":"Maas","orcid":"0000-0002-0845-1338","first_name":"Jan"},{"id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco","first_name":"Marco","orcid":"0000-0002-3242-7020","last_name":"Mondelli"}],"project":[{"_id":"fc31cba2-9c52-11eb-aca3-ff467d239cd2","name":"Taming Complexity in Partial Differential Systems","grant_number":"F6504"},{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"year":"2024","OA_type":"gold","file_date_updated":"2025-01-27T12:19:44Z","oa_version":"Published Version","quality_controlled":"1","date_published":"2024-06-01T00:00:00Z","title":"Improved convergence of score-based diffusion models via prediction-correction","abstract":[{"text":"Score-based generative models (SGMs) are powerful tools to sample from complex data distributions. Their underlying idea is to (i) run a forward process for time T1 by adding noise to the data, (ii) estimate its score function, and (iii) use such estimate to run a reverse process. As the reverse process is initialized with the stationary distribution of the forward one, the existing analysis paradigm requires T1→∞. This is however problematic: from a theoretical viewpoint, for a given precision of the score approximation, the convergence guarantee fails as T1 diverges; from a practical viewpoint, a large T1 increases computational costs and leads to error propagation. This paper addresses the issue by considering a version of the popular predictor-corrector scheme: after running the forward process, we first estimate the final distribution via an inexact Langevin dynamics and then revert the process. Our key technical contribution is to provide convergence guarantees which require to run the forward process only for a fixed finite time T1. Our bounds exhibit a mild logarithmic dependence on the input dimension and the subgaussian norm of the target distribution, have minimal assumptions on the data, and require only to control the L2 loss on the score approximation, which is the quantity minimized in practice.","lang":"eng"}],"publication_identifier":{"issn":["2835-8856"]},"date_updated":"2025-04-15T08:31:35Z","publication_status":"published","publication":"Transactions on Machine Learning Research","alternative_title":["TMLR"],"oa":1,"language":[{"iso":"eng"}],"OA_place":"publisher","acknowledgement":"Francesco Pedrotti and Jan Maas acknowledge support by the Austrian Science Fund (FWF) project 10.55776/F65. Marco Mondelli acknowledges support by the 2019 Lopez-Loreta prize.\r\n","arxiv":1,"article_processing_charge":"No","ddc":["000"],"_id":"18897","date_created":"2025-01-27T12:18:05Z","type":"conference","corr_author":"1"},{"year":"2024","OA_type":"green","oa_version":"Preprint","quality_controlled":"1","project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"author":[{"full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","last_name":"Bombari","first_name":"Simone"},{"last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"}],"citation":{"ista":"Bombari S, Mondelli M. 2024. How spurious features are memorized: Precise analysis for random and NTK features. 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 4267–4299.","apa":"Bombari, S., &#38; Mondelli, M. (2024). How spurious features are memorized: Precise analysis for random and NTK features. In <i>41st International Conference on Machine Learning</i> (Vol. 235, pp. 4267–4299). Vienna, Austria: ML Research Press.","ieee":"S. Bombari and M. Mondelli, “How spurious features are memorized: Precise analysis for random and NTK features,” in <i>41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 4267–4299.","chicago":"Bombari, Simone, and Marco Mondelli. “How Spurious Features Are Memorized: Precise Analysis for Random and NTK Features.” In <i>41st International Conference on Machine Learning</i>, 235:4267–99. ML Research Press, 2024.","ama":"Bombari S, Mondelli M. How spurious features are memorized: Precise analysis for random and NTK features. In: <i>41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:4267-4299.","short":"S. Bombari, M. Mondelli, in:, 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 4267–4299.","mla":"Bombari, Simone, and Marco Mondelli. “How Spurious Features Are Memorized: Precise Analysis for Random and NTK Features.” <i>41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 4267–99."},"department":[{"_id":"MaMo"}],"month":"07","external_id":{"arxiv":["2305.12100"]},"volume":235,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","conference":{"location":"Vienna, Austria","name":"ICML: International Conference on Machine Learning","end_date":"2024-07-27","start_date":"2024-07-21"},"day":"30","scopus_import":"1","publisher":"ML Research Press","status":"public","corr_author":"1","intvolume":"       235","arxiv":1,"article_processing_charge":"No","_id":"18972","date_created":"2025-01-30T07:29:47Z","page":"4267-4299","type":"conference","publication_status":"published","publication":"41st International Conference on Machine Learning","oa":1,"alternative_title":["PMLR"],"language":[{"iso":"eng"}],"OA_place":"repository","acknowledgement":"The authors were partially supported by the 2019 LopezLoreta prize, and they would like to thank (in alphabetical order) Grigorios Chrysos, Simone Maria Giancola, Mahyar\r\nJafari Nodeh, Christoph Lampert, Marco Miani, GuanWen Qiu, and Peter Sukenık for helpful discussions.","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2305.12100","open_access":"1"}],"date_published":"2024-07-30T00:00:00Z","title":"How spurious features are memorized: Precise analysis for random and NTK features","abstract":[{"text":"Deep learning models are known to overfit and memorize spurious features in the training dataset. While numerous empirical studies have aimed at understanding this phenomenon, a rigorous theoretical framework to quantify it is still missing. In this paper, we consider spurious features that are uncorrelated with the learning task, and we provide a precise characterization of how they are memorized via two separate terms: (i) the stability of the model with respect to individual training samples, and (ii) the feature alignment between the spurious pattern and the full sample. While the first term is well established in learning theory and it is connected to the generalization error in classical work, the second one is, to the best of our knowledge, novel. Our key technical result gives a precise characterization of the feature alignment for the two prototypical settings of random features (RF) and neural tangent kernel (NTK) regression. We prove that the memorization of spurious features weakens as the generalization capability increases and, through the analysis of the feature alignment, we unveil the role of the model and of its activation function. Numerical experiments show the predictive power of our theory on standard datasets (MNIST, CIFAR-10).","lang":"eng"}],"publication_identifier":{"eissn":["2640-3498"]},"date_updated":"2025-04-15T07:50:12Z"},{"publisher":"ML Research Press","scopus_import":"1","status":"public","day":"30","department":[{"_id":"MaMo"}],"month":"07","volume":235,"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","external_id":{"arxiv":["2402.02969"]},"conference":{"end_date":"2024-07-27","start_date":"2024-07-21","location":"Vienna, Austria","name":"ICML: International Conference on Machine Learning"},"citation":{"ista":"Bombari S, Mondelli M. 2024. Towards understanding the word sensitivity of attention layers: A study via random features. 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 4300–4328.","apa":"Bombari, S., &#38; Mondelli, M. (2024). Towards understanding the word sensitivity of attention layers: A study via random features. In <i>41st International Conference on Machine Learning</i> (Vol. 235, pp. 4300–4328). Vienna, Austria: ML Research Press.","ieee":"S. Bombari and M. Mondelli, “Towards understanding the word sensitivity of attention layers: A study via random features,” in <i>41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 4300–4328.","chicago":"Bombari, Simone, and Marco Mondelli. “Towards Understanding the Word Sensitivity of Attention Layers: A Study via Random Features.” In <i>41st International Conference on Machine Learning</i>, 235:4300–4328. ML Research Press, 2024.","ama":"Bombari S, Mondelli M. Towards understanding the word sensitivity of attention layers: A study via random features. In: <i>41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:4300-4328.","short":"S. Bombari, M. Mondelli, in:, 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 4300–4328.","mla":"Bombari, Simone, and Marco Mondelli. “Towards Understanding the Word Sensitivity of Attention Layers: A Study via Random Features.” <i>41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 4300–28."},"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"author":[{"last_name":"Bombari","first_name":"Simone","full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa"},{"orcid":"0000-0002-3242-7020","first_name":"Marco","last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco"}],"quality_controlled":"1","year":"2024","OA_type":"green","oa_version":"Preprint","publication_identifier":{"eissn":["2640-3498"]},"date_updated":"2025-04-15T07:50:12Z","date_published":"2024-07-30T00:00:00Z","title":"Towards understanding the word sensitivity of attention layers: A study via random features","abstract":[{"text":"Understanding the reasons behind the exceptional success of transformers requires a better analysis of why attention layers are suitable for NLP tasks. In particular, such tasks require predictive models to capture contextual meaning which often depends on one or few words, even if the sentence is long. Our work studies this key property, dubbed word sensitivity (WS), in the prototypical setting of random features. We show that attention layers enjoy high WS, namely, there exists a vector in the space of embeddings that largely perturbs the random attention features map. The argument critically exploits the role of the softmax in the attention layer, highlighting its benefit compared to other activations (e.g., ReLU). In contrast, the WS of standard random features is of order 1/n−−√, n being the number of words in the textual sample, and thus it decays with the length of the context. We then translate these results on the word sensitivity into generalization bounds: due to their low WS, random features provably cannot learn to distinguish between two sentences that differ only in a single word; in contrast, due to their high WS, random attention features have higher generalization capabilities. We validate our theoretical results with experimental evidence over the BERT-Base word embeddings of the imdb review dataset.","lang":"eng"}],"acknowledgement":"The authors were partially supported by the 2019 LopezLoreta prize, and they would like to thank Mohammad Hossein Amani, Lorenzo Beretta, and Clement Rebuffel for helpful discussions.","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2402.02969","open_access":"1"}],"publication_status":"published","publication":"41st International Conference on Machine Learning","oa":1,"alternative_title":["PMLR"],"language":[{"iso":"eng"}],"OA_place":"repository","date_created":"2025-01-30T07:35:49Z","page":"4300-4328","type":"conference","arxiv":1,"article_processing_charge":"No","_id":"18973","corr_author":"1","intvolume":"       235"},{"date_updated":"2026-04-07T13:00:02Z","related_material":{"record":[{"id":"18897","relation":"later_version","status":"public"},{"status":"public","id":"17336","relation":"dissertation_contains"}]},"doi":"10.48550/arXiv.2305.14164","status":"public","abstract":[{"lang":"eng","text":"Score-based generative models (SGMs) are powerful tools to sample from\r\ncomplex data distributions. Their underlying idea is to (i) run a forward\r\nprocess for time $T_1$ by adding noise to the data, (ii) estimate its score\r\nfunction, and (iii) use such estimate to run a reverse process. As the reverse\r\nprocess is initialized with the stationary distribution of the forward one, the\r\nexisting analysis paradigm requires $T_1\\to\\infty$. This is however\r\nproblematic: from a theoretical viewpoint, for a given precision of the score\r\napproximation, the convergence guarantee fails as $T_1$ diverges; from a\r\npractical viewpoint, a large $T_1$ increases computational costs and leads to\r\nerror propagation. This paper addresses the issue by considering a version of\r\nthe popular predictor-corrector scheme: after running the forward process, we\r\nfirst estimate the final distribution via an inexact Langevin dynamics and then\r\nrevert the process. Our key technical contribution is to provide convergence\r\nguarantees which require to run the forward process only for a fixed finite\r\ntime $T_1$. Our bounds exhibit a mild logarithmic dependence on the input\r\ndimension and the subgaussian norm of the target distribution, have minimal\r\nassumptions on the data, and require only to control the $L^2$ loss on the\r\nscore approximation, which is the quantity minimized in practice."}],"date_published":"2024-06-06T00:00:00Z","title":"Improved convergence of score-based diffusion models via prediction-correction","day":"06","external_id":{"arxiv":["2305.14164"]},"main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2305.14164"}],"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","department":[{"_id":"JaMa"},{"_id":"MaMo"}],"month":"06","language":[{"iso":"eng"}],"OA_place":"repository","citation":{"chicago":"Pedrotti, Francesco, Jan Maas, and Marco Mondelli. “Improved Convergence of Score-Based Diffusion Models via Prediction-Correction.” <i>ArXiv</i>, n.d. <a href=\"https://doi.org/10.48550/arXiv.2305.14164\">https://doi.org/10.48550/arXiv.2305.14164</a>.","ama":"Pedrotti F, Maas J, Mondelli M. Improved convergence of score-based diffusion models via prediction-correction. <i>arXiv</i>. doi:<a href=\"https://doi.org/10.48550/arXiv.2305.14164\">10.48550/arXiv.2305.14164</a>","ieee":"F. Pedrotti, J. Maas, and M. Mondelli, “Improved convergence of score-based diffusion models via prediction-correction,” <i>arXiv</i>. .","apa":"Pedrotti, F., Maas, J., &#38; Mondelli, M. (n.d.). Improved convergence of score-based diffusion models via prediction-correction. <i>arXiv</i>. <a href=\"https://doi.org/10.48550/arXiv.2305.14164\">https://doi.org/10.48550/arXiv.2305.14164</a>","ista":"Pedrotti F, Maas J, Mondelli M. Improved convergence of score-based diffusion models via prediction-correction. arXiv, <a href=\"https://doi.org/10.48550/arXiv.2305.14164\">10.48550/arXiv.2305.14164</a>.","mla":"Pedrotti, Francesco, et al. “Improved Convergence of Score-Based Diffusion Models via Prediction-Correction.” <i>ArXiv</i>, doi:<a href=\"https://doi.org/10.48550/arXiv.2305.14164\">10.48550/arXiv.2305.14164</a>.","short":"F. Pedrotti, J. Maas, M. Mondelli, ArXiv (n.d.)."},"publication":"arXiv","publication_status":"draft","oa":1,"type":"preprint","date_created":"2024-07-31T07:56:40Z","_id":"17350","arxiv":1,"author":[{"id":"d3ac8ac6-dc8d-11ea-abe3-e2a9628c4c3c","full_name":"Pedrotti, Francesco","first_name":"Francesco","last_name":"Pedrotti"},{"last_name":"Maas","first_name":"Jan","orcid":"0000-0002-0845-1338","full_name":"Maas, Jan","id":"4C5696CE-F248-11E8-B48F-1D18A9856A87"},{"orcid":"0000-0002-3242-7020","first_name":"Marco","last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","full_name":"Mondelli, Marco"}],"project":[{"_id":"fc31cba2-9c52-11eb-aca3-ff467d239cd2","grant_number":"F6504","name":"Taming Complexity in Partial Differential Systems"},{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"article_processing_charge":"No","corr_author":"1","oa_version":"Preprint","year":"2024"},{"oa_version":"Submitted Version","year":"2024","OA_type":"green","quality_controlled":"1","project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"},{"name":"Improving estimation and prediction of common complex disease risk","grant_number":"PCEGP3_181181","_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A"}],"author":[{"first_name":"Al","last_name":"Depope","id":"0b77531d-dbcd-11ea-9d1d-a8eee0bf3830","full_name":"Depope, Al"},{"last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"},{"full_name":"Robinson, Matthew Richard","id":"E5D42276-F5DA-11E9-8E24-6303E6697425","last_name":"Robinson","orcid":"0000-0001-8982-8813","first_name":"Matthew Richard"}],"isi":1,"citation":{"mla":"Depope, Al, et al. “Inference of Genetic Effects via Approximate Message Passing.” <i>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing</i>, IEEE, 2024, pp. 13151–55, doi:<a href=\"https://doi.org/10.1109/ICASSP48485.2024.10447198\">10.1109/ICASSP48485.2024.10447198</a>.","short":"A. Depope, M. Mondelli, M.R. Robinson, in:, 2024 IEEE International Conference on Acoustics, Speech, and Signal Processing, IEEE, 2024, pp. 13151–13155.","apa":"Depope, A., Mondelli, M., &#38; Robinson, M. R. (2024). Inference of genetic effects via approximate message passing. In <i>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing</i> (pp. 13151–13155). Seoul, Korea: IEEE. <a href=\"https://doi.org/10.1109/ICASSP48485.2024.10447198\">https://doi.org/10.1109/ICASSP48485.2024.10447198</a>","ama":"Depope A, Mondelli M, Robinson MR. Inference of genetic effects via approximate message passing. In: <i>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing</i>. IEEE; 2024:13151-13155. doi:<a href=\"https://doi.org/10.1109/ICASSP48485.2024.10447198\">10.1109/ICASSP48485.2024.10447198</a>","ieee":"A. Depope, M. Mondelli, and M. R. Robinson, “Inference of genetic effects via approximate message passing,” in <i>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing</i>, Seoul, Korea, 2024, pp. 13151–13155.","chicago":"Depope, Al, Marco Mondelli, and Matthew Richard Robinson. “Inference of Genetic Effects via Approximate Message Passing.” In <i>2024 IEEE International Conference on Acoustics, Speech, and Signal Processing</i>, 13151–55. IEEE, 2024. <a href=\"https://doi.org/10.1109/ICASSP48485.2024.10447198\">https://doi.org/10.1109/ICASSP48485.2024.10447198</a>.","ista":"Depope A, Mondelli M, Robinson MR. 2024. Inference of genetic effects via approximate message passing. 2024 IEEE International Conference on Acoustics, Speech, and Signal Processing. ICASSP: International Conference on Acoustics, Speech and Signal Processing, 13151–13155."},"external_id":{"isi":["001396233806078"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","conference":{"location":"Seoul, Korea","name":"ICASSP: International Conference on Acoustics, Speech and Signal Processing","end_date":"2024-04-19","start_date":"2024-04-14"},"department":[{"_id":"MaMo"},{"_id":"MaRo"}],"month":"04","day":"19","acknowledged_ssus":[{"_id":"ScienComp"}],"publisher":"IEEE","scopus_import":"1","status":"public","corr_author":"1","_id":"17147","article_processing_charge":"No","type":"conference","date_created":"2024-06-16T22:01:07Z","page":"13151-13155","language":[{"iso":"eng"}],"OA_place":"repository","publication_status":"published","publication":"2024 IEEE International Conference on Acoustics, Speech, and Signal Processing","oa":1,"main_file_link":[{"url":"https://openreview.net/forum?id=aQYCDxfZV0","open_access":"1"}],"acknowledgement":"This work was supported by a Lopez-Loreta Prize to MM, an SNSF Eccellenza Grant to MRR (PCEGP3-181181), and core funding from ISTA. The authors thank Philip Schniter, Matthew Stephens and Pragya Sur for valuable suggestions on an early version of the work. The authors acknowledge the participants and investigators of the UK Biobank study. High-performance\r\ncomputing was supported by the Scientific Service Units (SSU) of IST Austria through resources provided by Scientific Computing (SciComp).","abstract":[{"text":"Efficient utilization of large-scale biobank data is crucial for inferring the genetic basis of disease and predicting health outcomes from the DNA. Yet we lack efficient, accurate methods that scale to data where electronic health records are linked to whole genome sequence information. To address this issue, our paper develops a new algorithmic paradigm based on Approximate Message Passing (AMP), which is specifically tailored for genomic prediction and association testing. Our method yields comparable out-of-sample prediction accuracy to the state of the art on UK Biobank traits, whilst dramatically improving computational complexity, with a 8x-speed up in the run time. In addition, AMP theory provides a joint association testing framework, which outperforms the currently used REGENIE method, in roughly a third of the compute time. This first, truly large-scale application of the AMP framework lays the foundations for a far wider range of statistical analyses for hundreds of millions of variables measured on millions of people.","lang":"eng"}],"date_published":"2024-04-19T00:00:00Z","title":"Inference of genetic effects via approximate message passing","date_updated":"2026-07-13T14:57:55Z","doi":"10.1109/ICASSP48485.2024.10447198","publication_identifier":{"issn":["1520-6149"],"isbn":["9798350344851"]}},{"file_date_updated":"2024-10-05T22:30:05Z","oa_version":"Published Version","year":"2024","supervisor":[{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco"},{"last_name":"Alistarh","orcid":"0000-0003-3650-940X","first_name":"Dan-Adrian","full_name":"Alistarh, Dan-Adrian","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87"}],"author":[{"last_name":"Shevchenko","first_name":"Aleksandr","full_name":"Shevchenko, Aleksandr","id":"F2B06EC2-C99E-11E9-89F0-752EE6697425"}],"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"},{"grant_number":"W1260-N35","name":"Vienna Graduate School on Computational Optimization","_id":"9B9290DE-BA93-11EA-9121-9846C619BF3A"}],"degree_awarded":"PhD","citation":{"ieee":"A. Shevchenko, “High-dimensional limits in artificial neural networks,” Institute of Science and Technology Austria, 2024.","ama":"Shevchenko A. High-dimensional limits in artificial neural networks. 2024. doi:<a href=\"https://doi.org/10.15479/at:ista:17465\">10.15479/at:ista:17465</a>","chicago":"Shevchenko, Alexander. “High-Dimensional Limits in Artificial Neural Networks.” Institute of Science and Technology Austria, 2024. <a href=\"https://doi.org/10.15479/at:ista:17465\">https://doi.org/10.15479/at:ista:17465</a>.","apa":"Shevchenko, A. (2024). <i>High-dimensional limits in artificial neural networks</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/at:ista:17465\">https://doi.org/10.15479/at:ista:17465</a>","ista":"Shevchenko A. 2024. High-dimensional limits in artificial neural networks. Institute of Science and Technology Austria.","mla":"Shevchenko, Alexander. <i>High-Dimensional Limits in Artificial Neural Networks</i>. Institute of Science and Technology Austria, 2024, doi:<a href=\"https://doi.org/10.15479/at:ista:17465\">10.15479/at:ista:17465</a>.","short":"A. Shevchenko, High-Dimensional Limits in Artificial Neural Networks, Institute of Science and Technology Austria, 2024."},"user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","department":[{"_id":"GradSch"},{"_id":"DaAl"},{"_id":"MaMo"}],"month":"08","day":"29","acknowledged_ssus":[{"_id":"ScienComp"}],"has_accepted_license":"1","related_material":{"record":[{"relation":"part_of_dissertation","id":"11420","status":"public"},{"id":"14459","relation":"part_of_dissertation","status":"public"},{"status":"public","relation":"part_of_dissertation","id":"9198"},{"id":"17469","relation":"part_of_dissertation","status":"public"}]},"status":"public","file":[{"content_type":"application/pdf","date_created":"2024-09-02T09:23:32Z","date_updated":"2024-10-05T22:30:05Z","embargo":"2024-10-04","file_name":"thesis_a2b.pdf","creator":"ashevche","checksum":"da6dd3166078934577f6af93d27000e2","file_id":"17482","file_size":4468610,"access_level":"open_access","relation":"main_file"},{"relation":"source_file","content_type":"application/zip","date_created":"2024-09-02T09:23:46Z","date_updated":"2024-10-05T22:30:05Z","creator":"ashevche","embargo_to":"open_access","file_name":"Thesis Alex - ISTA.zip","file_id":"17483","checksum":"76a39ef252239560923cdda4ce0a31a4","file_size":15930999,"access_level":"closed"}],"publisher":"Institute of Science and Technology Austria","corr_author":"1","_id":"17465","ddc":["519"],"article_processing_charge":"No","type":"dissertation","date_created":"2024-08-28T15:14:25Z","page":"232","language":[{"iso":"eng"}],"OA_place":"repository","publication_status":"published","alternative_title":["ISTA Thesis"],"oa":1,"abstract":[{"lang":"eng","text":"In the modern age of machine learning, artificial neural networks have become an integral part\r\nof many practical systems. One of the key ingredients of the success of the deep learning\r\napproach is recent computational advances which allowed the training of models with billions\r\nof parameters on large-scale data. Such over-parameterized and data-hungry regimes pose a\r\nchallenge for the theoretical analysis of modern models since “classical” statistical wisdom\r\nis no longer applicable. In this view, it is paramount to extend or develop new machinery\r\nthat will allow tackling the neural network analysis under new challenging asymptotic regimes,\r\nwhich is the focus of this thesis.\r\nLarge neural network systems are usually optimized via “local” search algorithms, such\r\nas stochastic gradient descent (SGD). However, given the high-dimensional nature of the\r\nparameter space, it is a priori not clear why such a crude “local” approach works so remarkably\r\nwell in practice. We take a step towards demystifying this phenomenon by showing that\r\nthe landscape of the SGD training dynamics exhibits a few beneficial properties for the\r\noptimization. First, we show that along the SGD trajectory an over-parameterized network\r\nis dropout stable. The emergence of dropout stability allows to conclude that the minima\r\nfound by SGD are connected via a continuous path of small loss. This in turn means that\r\nthe high-dimensional landscape of the neural network optimization problem is provably not so\r\nunfavourable to gradient-based training, due to mode connectivity. Next, we show that SGD\r\nfor an over-parameterized network tends to find solutions that are functionally more “simple”.\r\nThis in turn means that the SGD minima are more robust, since a less complicated solution\r\nwill less likely overfit the data. More formally, for a prototypical example of a wide two-layer\r\nReLU network on a 1d regression task we show that the SGD algorithm is implicitly selective in\r\nits choice of an interpolating solution. Namely, at convergence the neural network implements\r\na piece-wise linear function with the number of linear regions depending only on the amount\r\nof training data. This is in contrast to a “smooth”-like behaviour which one would expect\r\ngiven such a severe over-parameterization of the model.\r\nDiverging from the generic supervised setting of classification and regression problems, we\r\nanalyze an auto-encoder model that is commonly used for representation learning and data\r\ncompression. Despite the wide applicability of the auto-encoding paradigm, the theoretical\r\nunderstanding of their behaviour is limited even in the simplistic shallow case. The related\r\nwork is restricted to extreme asymptotic regimes in which the auto-encoder is either severely\r\nover-parameterized or under-parameterized. In contrast, we provide a tight characterization\r\nfor the 1-bit compression of Gaussian signals in the challenging proportional regime, i.e., the\r\ninput dimension and the size of the compressed representation obey the same asymptotics.\r\nWe also show that gradient-based methods are able to find a globally optimal solution and\r\nthat the predictions made for Gaussian data extrapolate beyond - to the case of compression\r\nof natural images. Next, we relax the Gaussian assumption and study more structured input\r\nsources. We show that the shallow model is sometimes agnostic to the structure of the data\r\nvii\r\nwhich results in a Gaussian-like behaviour. We prove that making the decoding component\r\nslightly less shallow is already enough to escape the “curse” of Gaussian performance.\r\n"}],"date_published":"2024-08-29T00:00:00Z","title":"High-dimensional limits in artificial neural networks","date_updated":"2026-06-18T17:55:53Z","doi":"10.15479/at:ista:17465","publication_identifier":{"issn":["2663-337X"]}},{"quality_controlled":"1","oa_version":"Published Version","year":"2024","author":[{"full_name":"Kögler, Kevin","id":"94ec913c-dc85-11ea-9058-e5051ab2428b","last_name":"Kögler","first_name":"Kevin"},{"id":"F2B06EC2-C99E-11E9-89F0-752EE6697425","full_name":"Shevchenko, Aleksandr","first_name":"Aleksandr","last_name":"Shevchenko"},{"full_name":"Hassani, Hamed","last_name":"Hassani","first_name":"Hamed"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"conference":{"location":"Vienna, Austria","name":"ICML: International Conference on Machine Learning","end_date":"2024-07-27","start_date":"2024-07-21"},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","external_id":{"arxiv":["2402.05013"]},"volume":235,"month":"07","department":[{"_id":"DaAl"},{"_id":"MaMo"}],"citation":{"mla":"Kögler, Kevin, et al. “Compression of Structured Data with Autoencoders: Provable Benefit of Nonlinearities and Depth.” <i>Proceedings of the 41st International Conference on Machine Learning</i>, vol. 235, ML Research Press, 2024, pp. 24964–5015.","short":"K. Kögler, A. Shevchenko, H. Hassani, M. Mondelli, in:, Proceedings of the 41st International Conference on Machine Learning, ML Research Press, 2024, pp. 24964–25015.","chicago":"Kögler, Kevin, Alexander Shevchenko, Hamed Hassani, and Marco Mondelli. “Compression of Structured Data with Autoencoders: Provable Benefit of Nonlinearities and Depth.” In <i>Proceedings of the 41st International Conference on Machine Learning</i>, 235:24964–15. ML Research Press, 2024.","ama":"Kögler K, Shevchenko A, Hassani H, Mondelli M. Compression of structured data with autoencoders: Provable benefit of nonlinearities and depth. In: <i>Proceedings of the 41st International Conference on Machine Learning</i>. Vol 235. ML Research Press; 2024:24964-25015.","ieee":"K. Kögler, A. Shevchenko, H. Hassani, and M. Mondelli, “Compression of structured data with autoencoders: Provable benefit of nonlinearities and depth,” in <i>Proceedings of the 41st International Conference on Machine Learning</i>, Vienna, Austria, 2024, vol. 235, pp. 24964–25015.","apa":"Kögler, K., Shevchenko, A., Hassani, H., &#38; Mondelli, M. (2024). Compression of structured data with autoencoders: Provable benefit of nonlinearities and depth. In <i>Proceedings of the 41st International Conference on Machine Learning</i> (Vol. 235, pp. 24964–25015). Vienna, Austria: ML Research Press.","ista":"Kögler K, Shevchenko A, Hassani H, Mondelli M. 2024. Compression of structured data with autoencoders: Provable benefit of nonlinearities and depth. Proceedings of the 41st International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 235, 24964–25015."},"publisher":"ML Research Press","status":"public","scopus_import":"1","related_material":{"record":[{"status":"public","id":"17465","relation":"dissertation_contains"}]},"day":"01","intvolume":"       235","corr_author":"1","type":"conference","page":"24964-25015","date_created":"2024-08-29T11:47:57Z","ddc":["000"],"_id":"17469","article_processing_charge":"No","arxiv":1,"main_file_link":[{"open_access":"1","url":"https://proceedings.mlr.press/v235/kogler24a.html"}],"acknowledgement":"Kevin Kogler, Alexander Shevchenko and Marco Mondelli are supported by the 2019 Lopez-Loreta Prize. Hamed\r\nHassani acknowledges the support by the NSF CIF award (1910056) and the NSF Institute for CORE Emerging Methods in Data Science (EnCORE).","language":[{"iso":"eng"}],"alternative_title":["PMLR"],"oa":1,"publication":"Proceedings of the 41st International Conference on Machine Learning","publication_status":"published","date_updated":"2026-08-13T22:30:16Z","abstract":[{"lang":"eng","text":"Autoencoders are a prominent model in many empirical branches of machine learning and lossy data compression. However, basic theoretical questions remain unanswered even in a shallow two-layer setting. In particular, to what degree does a shallow autoencoder capture the structure of the underlying data distribution? For the prototypical case of the 1-bit compression of sparse Gaussian data, we prove that gradient descent converges to a solution that completely disregards the sparse structure of the input. Namely, the performance of the algorithm is the same as if it was compressing a Gaussian source - with no sparsity. For general data distributions, we give evidence of a phase transition phenomenon in the shape of the gradient descent minimizer, as a function of the data sparsity: below the critical sparsity level, the minimizer is a rotation taken uniformly at random (just like in the compression of non-sparse data); above the critical sparsity, the minimizer is the identity (up to a permutation). Finally, by exploiting a connection with approximate message passing algorithms, we show how to improve upon Gaussian performance for the compression of sparse data: adding a denoising function to a shallow architecture already reduces the loss provably, and a suitable multi-layer decoder leads to a further improvement. We validate our findings on image datasets, such as CIFAR-10 and MNIST."}],"title":"Compression of structured data with autoencoders: Provable benefit of nonlinearities and depth","date_published":"2024-07-01T00:00:00Z"},{"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"author":[{"full_name":"Bombari, Simone","id":"ca726dda-de17-11ea-bc14-f9da834f63aa","last_name":"Bombari","first_name":"Simone"},{"full_name":"Kiyani, Shayan","id":"f5a2b424-e339-11ed-8435-ff3b4fe70cf8","last_name":"Kiyani","first_name":"Shayan"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco"}],"year":"2023","oa_version":"Preprint","quality_controlled":"1","day":"27","publisher":"ML Research Press","status":"public","related_material":{"link":[{"url":"https://github.com/simone-bombari/beyond-universal-robustness","relation":"software"}]},"citation":{"ista":"Bombari S, Kiyani S, Mondelli M. 2023. Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. Proceedings of the 40th International Conference on Machine Learning. ICML: International Conference on Machine Learning, PMLR, vol. 202, 2738–2776.","ama":"Bombari S, Kiyani S, Mondelli M. Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. In: <i>Proceedings of the 40th International Conference on Machine Learning</i>. Vol 202. ML Research Press; 2023:2738-2776.","ieee":"S. Bombari, S. Kiyani, and M. Mondelli, “Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels,” in <i>Proceedings of the 40th International Conference on Machine Learning</i>, Honolulu, HI, United States, 2023, vol. 202, pp. 2738–2776.","chicago":"Bombari, Simone, Shayan Kiyani, and Marco Mondelli. “Beyond the Universal Law of Robustness: Sharper Laws for Random Features and Neural Tangent Kernels.” In <i>Proceedings of the 40th International Conference on Machine Learning</i>, 202:2738–76. ML Research Press, 2023.","apa":"Bombari, S., Kiyani, S., &#38; Mondelli, M. (2023). Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels. In <i>Proceedings of the 40th International Conference on Machine Learning</i> (Vol. 202, pp. 2738–2776). Honolulu, HI, United States: ML Research Press.","short":"S. Bombari, S. Kiyani, M. Mondelli, in:, Proceedings of the 40th International Conference on Machine Learning, ML Research Press, 2023, pp. 2738–2776.","mla":"Bombari, Simone, et al. “Beyond the Universal Law of Robustness: Sharper Laws for Random Features and Neural Tangent Kernels.” <i>Proceedings of the 40th International Conference on Machine Learning</i>, vol. 202, ML Research Press, 2023, pp. 2738–76."},"month":"10","department":[{"_id":"GradSch"},{"_id":"MaMo"}],"conference":{"end_date":"2023-07-29","start_date":"2023-07-23","location":"Honolulu, HI, United States","name":"ICML: International Conference on Machine Learning"},"volume":202,"external_id":{"arxiv":["2302.01629"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","article_processing_charge":"No","arxiv":1,"_id":"12859","page":"2738-2776","date_created":"2023-04-23T16:11:03Z","type":"conference","corr_author":"1","intvolume":"       202","title":"Beyond the universal law of robustness: Sharper laws for random features and neural tangent kernels","date_published":"2023-10-27T00:00:00Z","abstract":[{"text":"Machine learning models are vulnerable to adversarial perturbations, and a thought-provoking paper by Bubeck and Sellke has analyzed this phenomenon through the lens of over-parameterization: interpolating smoothly the data requires significantly more parameters than simply memorizing it. However, this \"universal\" law provides only a necessary condition for robustness, and it is unable to discriminate between models. In this paper, we address these gaps by focusing on empirical risk minimization in two prototypical settings, namely, random features and the neural tangent kernel (NTK). We prove that, for random features, the model is not robust for any degree of over-parameterization, even when the necessary condition coming from the universal law of robustness is satisfied. In contrast, for even activations, the NTK model meets the universal lower bound, and it is robust as soon as the necessary condition on over-parameterization is fulfilled. This also addresses a conjecture in prior work by Bubeck, Li and Nagaraj. Our analysis decouples the effect of the kernel of the model from an \"interaction matrix\", which describes the interaction with the test data and captures the effect of the activation. Our theoretical results are corroborated by numerical evidence on both synthetic and standard datasets (MNIST, CIFAR-10).","lang":"eng"}],"date_updated":"2025-04-15T07:50:16Z","alternative_title":["PMLR"],"oa":1,"publication_status":"published","publication":"Proceedings of the 40th International Conference on Machine Learning","language":[{"iso":"eng"}],"acknowledgement":"Simone Bombari and Marco Mondelli were partially supported by the 2019 Lopez-Loreta prize, and\r\nthe authors would like to thank Hamed Hassani for helpful discussions.\r\n","main_file_link":[{"url":"https://arxiv.org/abs/2302.01629","open_access":"1"}]},{"author":[{"full_name":"Barbier, Jean","last_name":"Barbier","first_name":"Jean"},{"last_name":"Camilli","first_name":"Francesco","full_name":"Camilli, Francesco"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","orcid":"0000-0002-3242-7020","first_name":"Marco"},{"last_name":"Sáenz","first_name":"Manuel","full_name":"Sáenz, Manuel"}],"project":[{"_id":"059876FA-7A3F-11EA-A408-12923DDC885E","name":"Prix Lopez-Loretta 2019 - Marco Mondelli"}],"isi":1,"file_date_updated":"2023-07-31T07:30:48Z","article_number":"e2302028120","oa_version":"Published Version","year":"2023","quality_controlled":"1","article_type":"original","day":"25","has_accepted_license":"1","related_material":{"link":[{"relation":"software","url":"https://github.com/fcamilli95/Structured-PCA-"}]},"tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"},"file":[{"relation":"main_file","success":1,"file_name":"2023_PNAS_Barbier.pdf","creator":"dernst","file_id":"13323","checksum":"1fc06228afdb3aa80cf8e7766bcf9dc5","access_level":"open_access","file_size":995933,"content_type":"application/pdf","date_created":"2023-07-31T07:30:48Z","date_updated":"2023-07-31T07:30:48Z"}],"scopus_import":"1","status":"public","publisher":"National Academy of Sciences","citation":{"ista":"Barbier J, Camilli F, Mondelli M, Sáenz M. 2023. Fundamental limits in structured principal component analysis and how to reach them. Proceedings of the National Academy of Sciences of the United States of America. 120(30), e2302028120.","ieee":"J. Barbier, F. Camilli, M. Mondelli, and M. Sáenz, “Fundamental limits in structured principal component analysis and how to reach them,” <i>Proceedings of the National Academy of Sciences of the United States of America</i>, vol. 120, no. 30. National Academy of Sciences, 2023.","chicago":"Barbier, Jean, Francesco Camilli, Marco Mondelli, and Manuel Sáenz. “Fundamental Limits in Structured Principal Component Analysis and How to Reach Them.” <i>Proceedings of the National Academy of Sciences of the United States of America</i>. National Academy of Sciences, 2023. <a href=\"https://doi.org/10.1073/pnas.2302028120\">https://doi.org/10.1073/pnas.2302028120</a>.","ama":"Barbier J, Camilli F, Mondelli M, Sáenz M. Fundamental limits in structured principal component analysis and how to reach them. <i>Proceedings of the National Academy of Sciences of the United States of America</i>. 2023;120(30). doi:<a href=\"https://doi.org/10.1073/pnas.2302028120\">10.1073/pnas.2302028120</a>","apa":"Barbier, J., Camilli, F., Mondelli, M., &#38; Sáenz, M. (2023). Fundamental limits in structured principal component analysis and how to reach them. <i>Proceedings of the National Academy of Sciences of the United States of America</i>. National Academy of Sciences. <a href=\"https://doi.org/10.1073/pnas.2302028120\">https://doi.org/10.1073/pnas.2302028120</a>","short":"J. Barbier, F. Camilli, M. Mondelli, M. Sáenz, Proceedings of the National Academy of Sciences of the United States of America 120 (2023).","mla":"Barbier, Jean, et al. “Fundamental Limits in Structured Principal Component Analysis and How to Reach Them.” <i>Proceedings of the National Academy of Sciences of the United States of America</i>, vol. 120, no. 30, e2302028120, National Academy of Sciences, 2023, doi:<a href=\"https://doi.org/10.1073/pnas.2302028120\">10.1073/pnas.2302028120</a>."},"user_id":"317138e5-6ab7-11ef-aa6d-ffef3953e345","volume":120,"external_id":{"isi":["001121663500001"],"pmid":["37463204"]},"department":[{"_id":"MaMo"}],"month":"07","ddc":["000"],"_id":"13315","article_processing_charge":"Yes (in subscription journal)","type":"journal_article","date_created":"2023-07-30T22:01:02Z","issue":"30","intvolume":"       120","abstract":[{"text":"How do statistical dependencies in measurement noise influence high-dimensional inference? To answer this, we study the paradigmatic spiked matrix model of principal components analysis (PCA), where a rank-one matrix is corrupted by additive noise. We go beyond the usual independence assumption on the noise entries, by drawing the noise from a low-order polynomial orthogonal matrix ensemble. The resulting noise correlations make the setting relevant for applications but analytically challenging. We provide characterization of the Bayes optimal limits of inference in this model. If the spike is rotation invariant, we show that standard spectral PCA is optimal. However, for more general priors, both PCA and the existing approximate message-passing algorithm (AMP) fall short of achieving the information-theoretic limits, which we compute using the replica method from statistical physics. We thus propose an AMP, inspired by the theory of adaptive Thouless–Anderson–Palmer equations, which is empirically observed to saturate the conjectured theoretical limit. This AMP comes with a rigorous state evolution analysis tracking its performance. Although we focus on specific noise distributions, our methodology can be generalized to a wide class of trace matrix ensembles at the cost of more involved expressions. Finally, despite the seemingly strong assumption of rotation-invariant noise, our theory empirically predicts algorithmic performance on real data, pointing at strong universality properties.","lang":"eng"}],"date_published":"2023-07-25T00:00:00Z","title":"Fundamental limits in structured principal component analysis and how to reach them","date_updated":"2025-09-09T12:41:50Z","doi":"10.1073/pnas.2302028120","publication_identifier":{"eissn":["1091-6490"]},"language":[{"iso":"eng"}],"publication_status":"published","publication":"Proceedings of the National Academy of Sciences of the United States of America","oa":1,"pmid":1,"acknowledgement":"J.B. was funded by the European Union (ERC, CHORAL, project number 101039794). Views and opinions expressed are however those of the author(s) only and do not necessarily reflect those of the European Union or the European Research Council. Neither the European Union nor the granting authority can be held responsible for them. M.M. was supported by the 2019 Lopez-Loreta Prize. We would like to thank the reviewers for the insightful comments and, in particular, for suggesting the BAMP-inspired denoisers leading to AMP-AP."},{"publication":"Proceedings of 2023 IEEE International Symposium on Information Theory","publication_status":"published","oa":1,"language":[{"iso":"eng"}],"acknowledgement":"The authors are partially supported by the 2019 Lopez-Loreta Prize. They would also like to thank Professor Jan Maas for providing valuable suggestions and comments on an early version of the work.","main_file_link":[{"url":"https://doi.org/10.48550/arXiv.2303.07245","open_access":"1"}],"date_published":"2023-06-30T00:00:00Z","title":"Concentration without independence via information measures","abstract":[{"text":"We propose a novel approach to concentration for non-independent random variables. The main idea is to ``pretend'' that the random variables are independent and pay a multiplicative price measuring how far they are from actually being independent. This price is encapsulated in the Hellinger integral between the joint and the product of the marginals, which is then upper bounded leveraging tensorisation properties. Our bounds represent a natural generalisation of concentration inequalities in the presence of dependence: we recover exactly the classical bounds (McDiarmid's inequality) when the random variables are independent. Furthermore, in a ``large deviations'' regime, we obtain the same decay in the probability as for the independent case, even when the random variables display non-trivial dependencies. To show this, we consider a number of applications of interest. First, we provide a bound for Markov chains with finite state space. Then, we consider the Simple Symmetric Random Walk, which is a non-contracting Markov chain, and a non-Markovian setting in which the stochastic process depends on its entire past. To conclude, we propose an application to Markov Chain Monte Carlo methods, where our approach leads to an improved lower bound on the minimum burn-in period required to reach a certain accuracy. In all of these settings, we provide a regime of parameters in which our bound fares better than what the state of the art can provide.","lang":"eng"}],"doi":"10.1109/isit54713.2023.10206899","publication_identifier":{"eissn":["2157-8117"],"eisbn":["9781665475549"]},"date_updated":"2025-09-04T13:06:52Z","corr_author":"1","arxiv":1,"article_processing_charge":"No","_id":"14922","date_created":"2024-02-02T11:18:40Z","page":"400-405","type":"conference","citation":{"mla":"Esposito, Amedeo Roberto, and Marco Mondelli. “Concentration without Independence via Information Measures.” <i>Proceedings of 2023 IEEE International Symposium on Information Theory</i>, IEEE, 2023, pp. 400–05, doi:<a href=\"https://doi.org/10.1109/isit54713.2023.10206899\">10.1109/isit54713.2023.10206899</a>.","short":"A.R. Esposito, M. Mondelli, in:, Proceedings of 2023 IEEE International Symposium on Information Theory, IEEE, 2023, pp. 400–405.","apa":"Esposito, A. R., &#38; Mondelli, M. (2023). Concentration without independence via information measures. In <i>Proceedings of 2023 IEEE International Symposium on Information Theory</i> (pp. 400–405). Taipei, Taiwan: IEEE. <a href=\"https://doi.org/10.1109/isit54713.2023.10206899\">https://doi.org/10.1109/isit54713.2023.10206899</a>","chicago":"Esposito, Amedeo Roberto, and Marco Mondelli. “Concentration without Independence via Information Measures.” In <i>Proceedings of 2023 IEEE International Symposium on Information Theory</i>, 400–405. IEEE, 2023. <a href=\"https://doi.org/10.1109/isit54713.2023.10206899\">https://doi.org/10.1109/isit54713.2023.10206899</a>.","ieee":"A. R. Esposito and M. Mondelli, “Concentration without independence via information measures,” in <i>Proceedings of 2023 IEEE International Symposium on Information Theory</i>, Taipei, Taiwan, 2023, pp. 400–405.","ama":"Esposito AR, Mondelli M. Concentration without independence via information measures. In: <i>Proceedings of 2023 IEEE International Symposium on Information Theory</i>. IEEE; 2023:400-405. doi:<a href=\"https://doi.org/10.1109/isit54713.2023.10206899\">10.1109/isit54713.2023.10206899</a>","ista":"Esposito AR, Mondelli M. 2023. Concentration without independence via information measures. Proceedings of 2023 IEEE International Symposium on Information Theory. ISIT: International Symposium on Information Theory, 400–405."},"department":[{"_id":"MaMo"}],"month":"06","user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","external_id":{"arxiv":["2303.07245"]},"conference":{"location":"Taipei, Taiwan","name":"ISIT: International Symposium on Information Theory","end_date":"2023-06-30","start_date":"2023-06-25"},"day":"30","related_material":{"record":[{"status":"public","relation":"later_version","id":"15172"}]},"publisher":"IEEE","scopus_import":"1","status":"public","year":"2023","oa_version":"Preprint","quality_controlled":"1","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"author":[{"first_name":"Amedeo Roberto","last_name":"Esposito","id":"9583e921-e1ad-11ec-9862-cef099626dc9","full_name":"Esposito, Amedeo Roberto"},{"last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020","full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425"}]},{"corr_author":"1","_id":"14924","ddc":["000"],"article_processing_charge":"No","arxiv":1,"type":"conference","date_created":"2024-02-02T11:21:56Z","language":[{"iso":"eng"}],"oa":1,"alternative_title":["TMLR"],"publication_status":"published","publication":"Transactions on Machine Learning Research","main_file_link":[{"open_access":"1","url":"https://doi.org/10.48550/arXiv.2210.06819"}],"acknowledgement":"D. Wu and M. Mondelli are partially supported by the 2019 Lopez-Loreta Prize. V. Kungurtsev was supported by the OP VVV project CZ.02.1.01/0.0/0.0/16_019/0000765 \"Research Center for Informatics\".","abstract":[{"text":"The stochastic heavy ball method (SHB), also known as stochastic gradient descent (SGD) with Polyak's momentum, is widely used in training neural networks. However, despite the remarkable success of such algorithm in practice, its theoretical characterization remains limited. In this paper, we focus on neural networks with two and three layers and provide a rigorous understanding of the properties of the solutions found by SHB: \\emph{(i)} stability after dropping out part of the neurons, \\emph{(ii)} connectivity along a low-loss path, and \\emph{(iii)} convergence to the global optimum.\r\nTo achieve this goal, we take a mean-field view and relate the SHB dynamics to a certain partial differential equation in the limit of large network widths. This mean-field perspective has inspired a recent line of work focusing on SGD while, in contrast, our paper considers an algorithm with momentum. More specifically, after proving existence and uniqueness of the limit differential equations, we show convergence to the global optimum and give a quantitative bound between the mean-field limit and the SHB dynamics of a finite-width network. Armed with this last bound, we are able to establish the dropout-stability and connectivity of SHB solutions.","lang":"eng"}],"title":"Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence","date_published":"2023-02-28T00:00:00Z","date_updated":"2026-06-18T17:41:36Z","oa_version":"Published Version","year":"2023","quality_controlled":"1","author":[{"id":"1a5914c2-896a-11ed-bdf8-fb80621a0635","full_name":"Wu, Diyuan","first_name":"Diyuan","last_name":"Wu"},{"first_name":"Vyacheslav","last_name":"Kungurtsev","full_name":"Kungurtsev, Vyacheslav"},{"full_name":"Mondelli, Marco","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli","first_name":"Marco","orcid":"0000-0002-3242-7020"}],"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"citation":{"ista":"Wu D, Kungurtsev V, Mondelli M. 2023. Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence. Transactions on Machine Learning Research. , TMLR, .","ieee":"D. Wu, V. Kungurtsev, and M. Mondelli, “Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence,” in <i>Transactions on Machine Learning Research</i>, 2023.","ama":"Wu D, Kungurtsev V, Mondelli M. Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence. In: <i>Transactions on Machine Learning Research</i>. ML Research Press; 2023.","chicago":"Wu, Diyuan, Vyacheslav Kungurtsev, and Marco Mondelli. “Mean-Field Analysis for Heavy Ball Methods: Dropout-Stability, Connectivity, and Global Convergence.” In <i>Transactions on Machine Learning Research</i>. ML Research Press, 2023.","apa":"Wu, D., Kungurtsev, V., &#38; Mondelli, M. (2023). Mean-field analysis for heavy ball methods: Dropout-stability, connectivity, and global convergence. In <i>Transactions on Machine Learning Research</i>. ML Research Press.","short":"D. Wu, V. Kungurtsev, M. Mondelli, in:, Transactions on Machine Learning Research, ML Research Press, 2023.","mla":"Wu, Diyuan, et al. “Mean-Field Analysis for Heavy Ball Methods: Dropout-Stability, Connectivity, and Global Convergence.” <i>Transactions on Machine Learning Research</i>, ML Research Press, 2023."},"external_id":{"arxiv":["2210.06819"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","month":"02","department":[{"_id":"MaMo"}],"day":"28","has_accepted_license":"1","publisher":"ML Research Press","status":"public","tmp":{"short":"CC BY (4.0)","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","image":"/images/cc_by.png"}}]
