[{"das_tickbox":"1","department":[{"_id":"GradSch"},{"_id":"MaRo"},{"_id":"MaMo"}],"day":"11","date_published":"2026-07-11T00:00:00Z","date_updated":"2026-07-28T07:08:15Z","status":"public","ddc":["576","610","006"],"page":"169","year":"2026","file_date_updated":"2026-07-13T14:56:41Z","publisher":"Institute of Science and Technology Austria","author":[{"full_name":"Depope, Al","first_name":"Al","id":"0b77531d-dbcd-11ea-9d1d-a8eee0bf3830","last_name":"Depope"}],"doi_confirm":"1","abstract":[{"text":"Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.\r\n\r\nFirst, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. \r\n\r\nSecond, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.\r\n\r\nCollectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.","lang":"eng"}],"_id":"22258","corr_author":"1","degree_awarded":"PhD","OA_place":"publisher","acknowledgement":"This work was supported in part by the Swiss National Science Foundation through the\r\nEccellenza Grant \"Improving estimation and prediction of common complex disease risk\"\r\n(grant number PCEGP3_181181); the European Research Council through the grant\r\n\"Inference in High Dimensions: Light-speed Algorithms and Information Limits\" (grant\r\nnumber 101161364); and the Fondation Jean-Jacques et Felicia Lopez-Loreta through the\r\nPrix Lopez-Loretta 2019.\r\n","file":[{"relation":"main_file","file_name":"2026_Depope_Al_Thesis.pdf","creator":"adepope","date_created":"2026-07-13T14:52:19Z","checksum":"9ab386790515628d957a194f30a7ccb4","file_id":"22316","content_type":"application/pdf","access_level":"open_access","file_size":25109878,"date_updated":"2026-07-13T14:52:19Z"},{"file_name":"2026_Depope_Al_Thesis.zip","relation":"source_file","content_type":"application/zip","access_level":"closed","file_size":1203199939,"date_updated":"2026-07-13T14:56:41Z","creator":"adepope","date_created":"2026-07-13T14:56:41Z","checksum":"8ed8fb63f76a695d5b6fec35343f4b90","file_id":"22317"}],"acknowledged_ssus":[{"_id":"ScienComp"}],"user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","publication_identifier":{"issn":["2663-337X"]},"doi":"10.15479/AT-ISTA-22258","date_created":"2026-07-10T13:27:20Z","keyword":["Approximate Message Passing","GWAS","Genomics","Proteomics","Survival modeling"],"type":"dissertation","publication_status":"published","supervisor":[{"orcid":"0000-0001-8982-8813","id":"E5D42276-F5DA-11E9-8E24-6303E6697425","last_name":"Robinson","full_name":"Robinson, Matthew Richard","first_name":"Matthew Richard"},{"id":"27EB676C-8706-11E9-9510-7717E6697425","orcid":"0000-0002-3242-7020","last_name":"Mondelli","first_name":"Marco","full_name":"Mondelli, Marco"}],"article_processing_charge":"No","project":[{"name":"Prix Lopez-Loretta 2019 - Marco Mondelli","_id":"059876FA-7A3F-11EA-A408-12923DDC885E"},{"name":"Inference in High Dimensions: Light-speed Algorithms and Information Limits","_id":"911e6d1f-16d5-11f0-9cad-c5c68c6a1cdf","grant_number":"101161364"},{"grant_number":"PCEGP3_181181","name":"Improving estimation and prediction of common complex disease risk","_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A"}],"citation":{"ama":"Depope A. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. 2026. doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>","short":"A. Depope, From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics, Institute of Science and Technology Austria, 2026.","ista":"Depope A. 2026. From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics. Institute of Science and Technology Austria.","apa":"Depope, A. (2026). <i>From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>","ieee":"A. Depope, “From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics,” Institute of Science and Technology Austria, 2026.","chicago":"Depope, Al. “From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics.” Institute of Science and Technology Austria, 2026. <a href=\"https://doi.org/10.15479/AT-ISTA-22258\">https://doi.org/10.15479/AT-ISTA-22258</a>.","mla":"Depope, Al. <i>From Sparse Selection to Risk Prediction: Approximate Message Passing for Proteomic Survival Models and Large-Scale Genomics</i>. Institute of Science and Technology Austria, 2026, doi:<a href=\"https://doi.org/10.15479/AT-ISTA-22258\">10.15479/AT-ISTA-22258</a>."},"related_material":{"record":[{"relation":"part_of_dissertation","status":"public","id":"21488"}]},"has_accepted_license":"1","alternative_title":["ISTA Thesis"],"oa_version":"Published Version","oa":1,"month":"07","title":"From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics","language":[{"iso":"eng"}]},{"citation":{"chicago":"Villanueva Marijuan, Ariadna. “Bayesian Linear Regression for Analyzing General Omics Data with Time-to-Event Phenotypes.” Institute of Science and Technology Austria, 2024. <a href=\"https://doi.org/10.15479/at:ista:17368\">https://doi.org/10.15479/at:ista:17368</a>.","mla":"Villanueva Marijuan, Ariadna. <i>Bayesian Linear Regression for Analyzing General Omics Data with Time-to-Event Phenotypes</i>. Institute of Science and Technology Austria, 2024, doi:<a href=\"https://doi.org/10.15479/at:ista:17368\">10.15479/at:ista:17368</a>.","ista":"Villanueva Marijuan A. 2024. Bayesian linear regression for analyzing general omics data with time-to-event phenotypes. Institute of Science and Technology Austria.","ieee":"A. Villanueva Marijuan, “Bayesian linear regression for analyzing general omics data with time-to-event phenotypes,” Institute of Science and Technology Austria, 2024.","apa":"Villanueva Marijuan, A. (2024). <i>Bayesian linear regression for analyzing general omics data with time-to-event phenotypes</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/at:ista:17368\">https://doi.org/10.15479/at:ista:17368</a>","short":"A. Villanueva Marijuan, Bayesian Linear Regression for Analyzing General Omics Data with Time-to-Event Phenotypes, Institute of Science and Technology Austria, 2024.","ama":"Villanueva Marijuan A. Bayesian linear regression for analyzing general omics data with time-to-event phenotypes. 2024. doi:<a href=\"https://doi.org/10.15479/at:ista:17368\">10.15479/at:ista:17368</a>"},"article_processing_charge":"No","supervisor":[{"id":"E5D42276-F5DA-11E9-8E24-6303E6697425","orcid":"0000-0001-8982-8813","last_name":"Robinson","first_name":"Matthew Richard","full_name":"Robinson, Matthew Richard"}],"publication_status":"published","language":[{"iso":"eng"}],"title":"Bayesian linear regression for analyzing general omics data with time-to-event phenotypes","month":"08","oa":1,"oa_version":"Published Version","has_accepted_license":"1","alternative_title":["ISTA Master's Thesis"],"publication_identifier":{"issn":["2791-4585"]},"user_id":"ba8df636-2132-11f1-aed0-ed93e2281fdd","type":"dissertation","license":"https://creativecommons.org/licenses/by-nc-sa/4.0/","keyword":["Epigenetics","Multi-omics","Bayesian regression"],"date_created":"2024-08-02T10:52:40Z","doi":"10.15479/at:ista:17368","abstract":[{"lang":"eng","text":"Recent advancements in molecular diagnostic techniques have enabled the collection of\r\nmultiple types of omics data from patients, including genomics, epigenomics, proteomics,\r\nand transcriptomics. However, we lack effective methods for integrating all these different\r\ndata types and combining them with clinical outcomes to study the molecular mechanisms\r\nthat govern pathological phenotypes. We present multi-omics BayesW, a penalized Bayesian\r\nregression method that can handle general omics data for survival analysis of time-to-event\r\nphenotypes. Our method can: (1) accommodate incomplete data by allowing censored\r\nindividuals, (2) use continuous time-to-event data to test associations of markers with a\r\nphenotype and (3) estimate effects jointly while allowing for independent groups of biological\r\nmarkers. Extensive simulations using planted signals on real data demonstrate that our model\r\naccurately retrieves the true parameters of the model while controlling for false discoveries\r\nand maintaining the expected prediction accuracy. We address data correlations by estimating\r\nthe effects jointly, even between omic groups, while also estimating the individual variance\r\nexplained by each group. We apply our model to two datasets. Using 18,000 individuals from\r\nthe Generation Scotland study we model the association of time at onset of Type 2 Diabetes,\r\nStroke, Ischemic Disease, and Osteoarthritis from baseline study entry, with 831,724 CpG\r\nmethylation probes. We find that large proportions of variation in disease onset times can\r\nbe attributed to methylation as measured in whole blood at baseline in individuals without\r\ndisease symptoms. We then apply our model to The Cancer Genome Atlas (TCGA) pan-cancer\r\ndataset, in which we use 5 types of omics: copy number variation, epigenetics, somatic\r\nmutations, miRNA, and gene expression. For cancer survival age-at-onset we find that, when\r\nfitting the 5 groups together, almost all variation attributable to \"omics\" data is explained by\r\nDNA methylation. When considering progression times, both methylation and gene expression\r\nexplain a large part of the variance. We found 2 genes that are significantly associated (95%\r\nposterior inclusion probability) with cancer survival time, conditional on all other genome-wide\r\nomics data variation. Owing to the vast variability of mechanisms characterizing different\r\ncancers, there are likely few specific genes with a strong signal in a pan-cancer setting. Taken\r\ntogether, we showed the applicability of our multi-omics BayesW model to a wide-range of\r\nbiological questions in multi-omics data.\r\n"}],"author":[{"full_name":"Villanueva Marijuan, Ariadna","first_name":"Ariadna","last_name":"Villanueva Marijuan","id":"e0ae4864-133f-11ed-8f02-adaa8dd27540"}],"publisher":"Institute of Science and Technology Austria","file_date_updated":"2025-02-14T23:30:03Z","year":"2024","page":"60","file":[{"file_size":13052436,"date_updated":"2025-02-14T23:30:03Z","access_level":"open_access","content_type":"application/pdf","file_id":"17433","checksum":"0c2daa174609f0c00919dccc5701d375","embargo":"2025-02-14","creator":"avillanu","date_created":"2024-08-14T11:51:24Z","file_name":"Masters_thesis_AriadnaVillanueva.pdf","relation":"main_file"},{"relation":"source_file","file_name":"Masters thesis-AriadnaVillanueva.zip","checksum":"e9ed4465dfa539ac4c3a8d4d0b6271a1","date_created":"2024-08-14T11:51:57Z","creator":"avillanu","file_id":"17434","access_level":"closed","content_type":"application/zip","embargo_to":"open_access","date_updated":"2025-02-14T23:30:03Z","file_size":45642547}],"OA_place":"publisher","degree_awarded":"MS","corr_author":"1","_id":"17368","department":[{"_id":"GradSch"},{"_id":"MaRo"}],"ddc":["610"],"status":"public","tmp":{"name":"Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International (CC BY-NC-SA 4.0)","short":"CC BY-NC-SA (4.0)","legal_code_url":"https://creativecommons.org/licenses/by-nc-sa/4.0/legalcode","image":"/images/cc_by_nc_sa.png"},"date_updated":"2026-04-07T13:03:41Z","date_published":"2024-08-13T00:00:00Z","day":"13"},{"abstract":[{"text":"This thesis consists of two pieces of work in the broader field of computational biology,\r\nboth of which are methods for the analysis of large scale biological data, implemented in\r\nefficient software.\r\nChapter 2 introduces a statistical software for causal discovery and inference from observed\r\ngenetic marker and phenotypic trait data. We explore in simulation how well the method\r\ncan fine-map genetic effects, find the correct causal structure among tens of traits and\r\nmillions of genetic markers, and infer the causal effect size for the discovered causal\r\nrelations. We then apply the method to 8 million markers and 17 traits from the UK\r\nBiobank and show that many relationships found with other methods are likely due to\r\nthe effects of hidden confounders.\r\nChapter 3 describes how this method can be applied to longitudinal data. I show how one\r\ncan incorporate the background knowledge present in the known order of measurements to\r\nimprove the accuracy of the causal discovery process, and explore the method’s ability to\r\nidentify age specific genetic effects, and how the error rates of this recovery are influenced\r\nby missing data due to different censoring mechanisms.\r\nChapter 4 introduces a statistical software for the comparison of chromatin contact maps\r\nbased on the structural similarity index. We explore the robustness of the method to\r\nnoise and size differences of the compared maps, show how it can measure evolutionary\r\nconservation of topological features by providing a similarity ranking of syntenic regions,\r\nand finally how it can detect alterations in 3D genome structure due to genetic mutations\r\nin samples of medical relevance.\r\n","lang":"eng"}],"doi_confirm":"1","author":[{"first_name":"Nick N","full_name":"Machnik, Nick N","id":"3591A0AA-F248-11E8-B48F-1D18A9856A87","orcid":"0000-0001-6617-9742","last_name":"Machnik"}],"publisher":"Institute of Science and Technology Austria","file_date_updated":"2025-06-12T22:30:02Z","page":"138","year":"2024","acknowledgement":"I would like to thank the Swiss National Science Foundation for funding parts of this work\r\nthrough the Eccellenza Grant \"Improving estimation and prediction of common complex\r\ndisease risk\" with grant number PCEGP3_181181.","file":[{"relation":"main_file","file_name":"NickMachnikThesisFinal_pdfa_conv.pdf","embargo":"2025-06-12","creator":"nmachnik","date_created":"2024-12-11T11:59:54Z","checksum":"d45e4d170f9a70a1f69b44b99bd058e4","file_id":"18649","content_type":"application/pdf","access_level":"open_access","date_updated":"2025-06-12T22:30:02Z","file_size":12845009},{"file_name":"thesis.zip","relation":"source_file","date_updated":"2025-06-12T22:30:02Z","embargo_to":"open_access","file_size":14189810,"access_level":"closed","content_type":"application/zip","file_id":"18650","checksum":"f88c9acc62002395ec4dcbdb5eea8b82","date_created":"2024-12-11T11:59:34Z","creator":"nmachnik"}],"OA_place":"publisher","corr_author":"1","degree_awarded":"PhD","_id":"18642","department":[{"_id":"GradSch"},{"_id":"MaRo"}],"ddc":["576"],"status":"public","date_updated":"2026-07-29T13:02:44Z","date_published":"2024-12-11T00:00:00Z","day":"11","citation":{"mla":"Machnik, Nick N. <i>Algorithms for Causal Learning and Comparative Analysis for Genomic Data</i>. Institute of Science and Technology Austria, 2024, doi:<a href=\"https://doi.org/10.15479/at:ista:18642\">10.15479/at:ista:18642</a>.","chicago":"Machnik, Nick N. “Algorithms for Causal Learning and Comparative Analysis for Genomic Data.” Institute of Science and Technology Austria, 2024. <a href=\"https://doi.org/10.15479/at:ista:18642\">https://doi.org/10.15479/at:ista:18642</a>.","apa":"Machnik, N. N. (2024). <i>Algorithms for causal learning and comparative analysis for genomic data</i>. Institute of Science and Technology Austria. <a href=\"https://doi.org/10.15479/at:ista:18642\">https://doi.org/10.15479/at:ista:18642</a>","ieee":"N. N. Machnik, “Algorithms for causal learning and comparative analysis for genomic data,” Institute of Science and Technology Austria, 2024.","ista":"Machnik NN. 2024. Algorithms for causal learning and comparative analysis for genomic data. Institute of Science and Technology Austria.","short":"N.N. Machnik, Algorithms for Causal Learning and Comparative Analysis for Genomic Data, Institute of Science and Technology Austria, 2024.","ama":"Machnik NN. Algorithms for causal learning and comparative analysis for genomic data. 2024. doi:<a href=\"https://doi.org/10.15479/at:ista:18642\">10.15479/at:ista:18642</a>"},"project":[{"_id":"9B8D11D6-BA93-11EA-9121-9846C619BF3A","name":"Improving estimation and prediction of common complex disease risk","grant_number":"PCEGP3_181181"}],"article_processing_charge":"No","publication_status":"published","supervisor":[{"first_name":"Matthew Richard","full_name":"Robinson, Matthew Richard","last_name":"Robinson","id":"E5D42276-F5DA-11E9-8E24-6303E6697425","orcid":"0000-0001-8982-8813"}],"language":[{"iso":"eng"}],"title":"Algorithms for causal learning and comparative analysis for genomic data","month":"12","oa":1,"oa_version":"Published Version","alternative_title":["ISTA Thesis"],"has_accepted_license":"1","related_material":{"record":[{"id":"18648","status":"public","relation":"part_of_dissertation"},{"relation":"part_of_dissertation","id":"8707","status":"public"}]},"publication_identifier":{"issn":["2663-337X"]},"user_id":"8b945eb4-e2f2-11eb-945a-df72226e66a9","type":"dissertation","date_created":"2024-12-10T13:49:15Z","doi":"10.15479/at:ista:18642"}]
