[{"citation":{"short":"A. Depope, From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics, Institute of Science and Technology Austria, 2026.","ama":"Depope A. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. 2026. doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>","chicago":"Depope, Al. “From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics.” Institute of Science and Technology Austria, 2026. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>.","mla":"Depope, Al. <i>From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics</i>. Institute of Science and Technology Austria, 2026, doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>.","ista":"Depope A. 2026. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. Institute of Science and Technology Austria.","apa":"Depope, A. (2026). <i>From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>","ieee":"A. Depope, “From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics,” Institute of Science and Technology Austria, 2026."},"project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"},{"name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf","grant_number":"101161364"},{"grant_number":"PCEGP3_181181","name":"Improving estimation and prediction of common complex disease risk","_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A"}],"article_processing_charge":"No","supervisor":[{"full_name":"Robinson, Matthew Richard","first_name":"Matthew Richard","last_name":"Robinson","orcid":"0000-0001-8982-8813","id":"E5D42276-F5DA-11E9-8E24-6303E6697425"},{"full_name":"Mondelli, Marco","first_name":"Marco","orcid":"0000-0002-3242-7020","id":"27EB676C-8706-11E9-9510-7717E6697425","last_name":"Mondelli"}],"publication_status":"published","language":[{"iso":"eng"}],"month":"07","title":"From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics","oa":1,"oa_version":"Published Version","alternative_title":["ISTA Thesis"],"has_accepted_license":"1","related_material":{"record":[{"relation":"part_of_dissertation","status":"public","id":"21488"}]},"publication_identifier":{"issn":["2663-337X"]},"user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","acknowledged_ssus":[{"_id":"ScienComp"}],"type":"dissertation","keyword":["Approximate Message Passing","GWAS","Genomics","Proteomics","Survival modeling"],"date_created":"2026-07-10T13:27:20Z","doi":"10.15479/AT-ISTA-22258","abstract":[{"text":"Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.\r\n\r\nFirst, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. \r\n\r\nSecond, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.\r\n\r\nCollectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.","lang":"eng"}],"doi_confirm":"1","author":[{"last_name":"Depope","id":"0b77531d-dbcd-11ea-9d1d-a8eee0bf3830","first_name":"Al","full_name":"Depope, Al"}],"publisher":"Institute of Science and Technology Austria","file_date_updated":"2026-07-13T14:56:41Z","year":"2026","page":"169","file":[{"file_size":25109878,"date_updated":"2026-07-13T14:52:19Z","access_level":"open_access","content_type":"application/pdf","file_id":"22316","checksum":"9ab386790515628d957a194f30a7ccb4","date_created":"2026-07-13T14:52:19Z","creator":"adepope","file_name":"2026_Depope_Al_Thesis.pdf","relation":"main_file"},{"file_name":"2026_Depope_Al_Thesis.zip","relation":"source_file","content_type":"application/zip","access_level":"closed","date_updated":"2026-07-13T14:56:41Z","file_size":1203199939,"creator":"adepope","date_created":"2026-07-13T14:56:41Z","checksum":"8ed8fb63f76a695d5b6fec35343f4b90","file_id":"22317"}],"acknowledgement":"This work was supported in part by the Swiss National Science Foundation through the\r\nEccellenza Grant \"Improving estimation and prediction of common complex disease risk\"\r\n(grant number PCEGP3_181181); the European Research Council through the grant\r\n\"Inference in High Dimensions: Light-speed Algorithms and Information Limits\" (grant\r\nnumber 101161364); and the Fondation Jean-Jacques et Felicia Lopez-Loreta through the\r\nPrix Lopez-Loretta 2019.\r\n","OA_place":"publisher","degree_awarded":"PhD","corr_author":"1","_id":"22258","department":[{"_id":"GradSch"},{"_id":"MaRo"},{"_id":"MaMo"}],"das_tickbox":"1","ddc":["576","610","006"],"status":"public","date_updated":"2026-07-28T07:08:15Z","date_published":"2026-07-11T00:00:00Z","day":"11"},{"file_date_updated":"2026-07-01T06:22:15Z","OA_type":"hybrid","year":"2026","page":"411-439","author":[{"full_name":"Zhang, Yihan","first_name":"Yihan","last_name":"Zhang"},{"last_name":"Mondelli","id":"27EB676C-8706-11E9-9510-7717E6697425","orcid":"0000-0002-3242-7020","first_name":"Marco","full_name":"Mondelli, Marco"},{"last_name":"Venkataramanan","full_name":"Venkataramanan, Ramji","first_name":"Ramji"}],"abstract":[{"text":"In a mixed generalized linear model, the goal is to learn multiple signals from unlabeled observations: each sample comes from exactly one signal, but it is not known which one. We consider the prototypical problem of estimating two statistically independent signals in a mixed generalized linear model with Gaussian covariates. Spectral methods are a popular class of estimators which output the top two eigenvectors of a suitable data-dependent matrix. However, despite the wide applicability, their design is still obtained via heuristic considerations, and the number of samples 𝑛 needed to guarantee recovery is superlinear in the signal dimension 𝑑. In this paper, we develop exact asymptotics on spectral methods in the challenging proportional regime in which 𝑛,𝑑 grow large and their ratio converges to a finite constant. This allows us optimize the design of the spectral method, and combine it with a simple linear estimator, to minimize the estimation error. Our characterization exploits a mix of tools from random matrices, free probability, and the theory of approximate message passing algorithms. Numerical simulations for mixed linear regression and phase retrieval demonstrate the advantage enabled by our analysis over existing designs of spectral methods.","lang":"eng"}],"publisher":"SIAM","_id":"22228","OA_place":"publisher","file":[{"success":1,"relation":"main_file","file_name":"2026_SIAMJourmathDataScience_Zhang.pdf","creator":"dernst","date_created":"2026-07-01T06:22:15Z","checksum":"5cfd350dc64d1476063e959316dbff65","file_id":"22230","content_type":"application/pdf","access_level":"open_access","file_size":1210346,"date_updated":"2026-07-01T06:22:15Z"}],"acknowledgement":"The first and second authors were partially supported by the 2019 Lopez-Loreta prize.","corr_author":"1","das_tickbox":"0","department":[{"_id":"MaMo"}],"publication":"SIAM Journal on Mathematics of Data Science","date_published":"2026-06-01T00:00:00Z","intvolume":"         8","day":"01","status":"public","tmp":{"short":"CC BY (4.0)","legal_code_url":"https://creativecommons.org/licenses/by/4.0/legalcode","name":"Creative Commons Attribution 4.0 International Public License (CC-BY 4.0)","image":"/images/cc_by.png"},"ddc":["000"],"date_updated":"2026-08-12T06:23:07Z","external_id":{"arxiv":["2211.11368"]},"publication_status":"published","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"}],"citation":{"ista":"Zhang Y, Mondelli M, Venkataramanan R. 2026. Precise asymptotics for spectral methods in mixed generalized linear models. SIAM Journal on Mathematics of Data Science. 8(2), 411–439.","apa":"Zhang, Y., Mondelli, M., &#38; Venkataramanan, R. (2026). Precise asymptotics for spectral methods in mixed generalized linear models. <i>SIAM Journal on Mathematics of Data Science</i>. SIAM. <a href=\"https://doi.org/10.1137/24m1702854\">https://doi.org/10.1137/24m1702854</a>","ieee":"Y. Zhang, M. Mondelli, and R. Venkataramanan, “Precise asymptotics for spectral methods in mixed generalized linear models,” <i>SIAM Journal on Mathematics of Data Science</i>, vol. 8, no. 2. SIAM, pp. 411–439, 2026.","chicago":"Zhang, Yihan, Marco Mondelli, and Ramji Venkataramanan. “Precise Asymptotics for Spectral Methods in Mixed Generalized Linear Models.” <i>SIAM Journal on Mathematics of Data Science</i>. SIAM, 2026. <a href=\"https://doi.org/10.1137/24m1702854\">https://doi.org/10.1137/24m1702854</a>.","mla":"Zhang, Yihan, et al. “Precise Asymptotics for Spectral Methods in Mixed Generalized Linear Models.” <i>SIAM Journal on Mathematics of Data Science</i>, vol. 8, no. 2, SIAM, 2026, pp. 411–39, doi:<a href=\"https://doi.org/10.1137/24m1702854\">10.1137/24m1702854</a>.","ama":"Zhang Y, Mondelli M, Venkataramanan R. Precise asymptotics for spectral methods in mixed generalized linear models. <i>SIAM Journal on Mathematics of Data Science</i>. 2026;8(2):411-439. doi:<a href=\"https://doi.org/10.1137/24m1702854\">10.1137/24m1702854</a>","short":"Y. Zhang, M. Mondelli, R. Venkataramanan, SIAM Journal on Mathematics of Data Science 8 (2026) 411–439."},"article_processing_charge":"Yes (in subscription journal)","quality_controlled":"1","issue":"2","oa_version":"Published Version","oa":1,"has_accepted_license":"1","language":[{"iso":"eng"}],"supplementarymaterial":"no","month":"06","title":"Precise asymptotics for spectral methods in mixed generalized linear models","researchdata_availability":"no","PlanS_conform":"1","publication_identifier":{"eissn":["2577-0187"]},"user_id":"2DF688A6-F248-11E8-B48F-1D18A9856A87","mathsc":["62E20","62J05","62J12"],"date_created":"2026-06-30T13:03:41Z","keyword":["spectral estimator","generalized linear models","mixed regression","high-dimensional asymptotics","random matrix theory","approximate message passing (AMP)"],"doi":"10.1137/24m1702854","type":"journal_article","volume":8,"arxiv":1,"scopus_import":"1","article_type":"original"}]
