@phdthesis{22258,
  abstract     = {Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.

First, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. 

Second, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.

Collectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.},
  author       = {Depope, Al},
  issn         = {2663-337X},
  keywords     = {Approximate Message Passing, GWAS, Genomics, Proteomics, Survival modeling},
  pages        = {169},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics}},
  doi          = {10.15479/AT-ISTA-22258},
  year         = {2026},
}

@phdthesis{17368,
  abstract     = {Recent advancements in molecular diagnostic techniques have enabled the collection of
multiple types of omics data from patients, including genomics, epigenomics, proteomics,
and transcriptomics. However, we lack effective methods for integrating all these different
data types and combining them with clinical outcomes to study the molecular mechanisms
that govern pathological phenotypes. We present multi-omics BayesW, a penalized Bayesian
regression method that can handle general omics data for survival analysis of time-to-event
phenotypes. Our method can: (1) accommodate incomplete data by allowing censored
individuals, (2) use continuous time-to-event data to test associations of markers with a
phenotype and (3) estimate effects jointly while allowing for independent groups of biological
markers. Extensive simulations using planted signals on real data demonstrate that our model
accurately retrieves the true parameters of the model while controlling for false discoveries
and maintaining the expected prediction accuracy. We address data correlations by estimating
the effects jointly, even between omic groups, while also estimating the individual variance
explained by each group. We apply our model to two datasets. Using 18,000 individuals from
the Generation Scotland study we model the association of time at onset of Type 2 Diabetes,
Stroke, Ischemic Disease, and Osteoarthritis from baseline study entry, with 831,724 CpG
methylation probes. We find that large proportions of variation in disease onset times can
be attributed to methylation as measured in whole blood at baseline in individuals without
disease symptoms. We then apply our model to The Cancer Genome Atlas (TCGA) pan-cancer
dataset, in which we use 5 types of omics: copy number variation, epigenetics, somatic
mutations, miRNA, and gene expression. For cancer survival age-at-onset we find that, when
fitting the 5 groups together, almost all variation attributable to "omics" data is explained by
DNA methylation. When considering progression times, both methylation and gene expression
explain a large part of the variance. We found 2 genes that are significantly associated (95%
posterior inclusion probability) with cancer survival time, conditional on all other genome-wide
omics data variation. Owing to the vast variability of mechanisms characterizing different
cancers, there are likely few specific genes with a strong signal in a pan-cancer setting. Taken
together, we showed the applicability of our multi-omics BayesW model to a wide-range of
biological questions in multi-omics data.
},
  author       = {Villanueva Marijuan, Ariadna},
  issn         = {2791-4585},
  keywords     = {Epigenetics, Multi-omics, Bayesian regression},
  pages        = {60},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Bayesian linear regression for analyzing general omics data with time-to-event phenotypes}},
  doi          = {10.15479/at:ista:17368},
  year         = {2024},
}

@phdthesis{18642,
  abstract     = {This thesis consists of two pieces of work in the broader field of computational biology,
both of which are methods for the analysis of large scale biological data, implemented in
efficient software.
Chapter 2 introduces a statistical software for causal discovery and inference from observed
genetic marker and phenotypic trait data. We explore in simulation how well the method
can fine-map genetic effects, find the correct causal structure among tens of traits and
millions of genetic markers, and infer the causal effect size for the discovered causal
relations. We then apply the method to 8 million markers and 17 traits from the UK
Biobank and show that many relationships found with other methods are likely due to
the effects of hidden confounders.
Chapter 3 describes how this method can be applied to longitudinal data. I show how one
can incorporate the background knowledge present in the known order of measurements to
improve the accuracy of the causal discovery process, and explore the method’s ability to
identify age specific genetic effects, and how the error rates of this recovery are influenced
by missing data due to different censoring mechanisms.
Chapter 4 introduces a statistical software for the comparison of chromatin contact maps
based on the structural similarity index. We explore the robustness of the method to
noise and size differences of the compared maps, show how it can measure evolutionary
conservation of topological features by providing a similarity ranking of syntenic regions,
and finally how it can detect alterations in 3D genome structure due to genetic mutations
in samples of medical relevance.
},
  author       = {Machnik, Nick N},
  issn         = {2663-337X},
  pages        = {138},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Algorithms for causal learning and comparative analysis for genomic data}},
  doi          = {10.15479/at:ista:18642},
  year         = {2024},
}

