@article{21503,
  abstract     = {Currently, pharmacogenetics relies on partially annotated star alleles, leaving novel variants and complex haplotypes uninterpretable. Computational scoring frameworks could overcome these limitations. Here, we comprehensively evaluated the ability of existing (CADD, FATHMM-XF, PROVEAN, MutationAssessor, SIFT, PhyloP100, APF, APF2) and novel (PharmGScore and PharmMLScore) variant effect predictors to assess pharmacogenetic alleles in multiple scenarios. Altogether we analyzed 541 PharmVar alleles, high‑throughput CYP2C9 and CYP2C19 mutational maps, and 200 642 UK Biobank exomes linked with health records containing antidepressant treatment outcomes. Many evaluated tools, especially ensemble frameworks, matched or exceeded star allele classifications (ROC‑AUC up to 0.85 for allele definitions, 0.95 in vitro; TPR up to 0.99 for exomes) and accurately predicted severe antidepressant adverse events for carriers of deleterious variants in CYP2C19 (OR 1.20–1.35). Our findings show that computational predictors deliver star allele accuracy while overcoming their limitations. With additional validation, computational tools could enhance clinical decision frameworks by enabling continuous scoring, incorporating previously unknown variants, and providing genome-wide applicability.},
  author       = {Hajto, Jacek and Piechota, Marcin and Krätschmer, Ilse and Konowalska, Paula and Boyle, Gabriel E. and Fowler, Douglas M. and Borczyk, Malgorzata and Korostynski, Michal},
  issn         = {1473-1150},
  journal      = {Pharmacogenomics Journal},
  number       = {2},
  publisher    = {Springer Nature},
  title        = {{Computational variant predictors for pharmacogenomics: From evaluation of single alleles to assessment of adverse drug reactions to antidepressants}},
  doi          = {10.1038/s41397-026-00399-0},
  volume       = {26},
  year         = {2026},
}

@article{21484,
  abstract     = {An individual's phenotype reflects a complex interplay of the direct effects of their DNA, epigenetic modifications of their DNA induced by their parents, and indirect effects of their parents' DNA. Here, we derive how the genetic variance within a population is changed under the influence of indirect maternal, paternal and parent-of-origin effects under random mating. We also consider indirect effects of a sibling, in particular how the genetic variance is altered when looking at the phenotypic difference between two siblings. The calculations are then extended to include assortative mating (AM), which alters the variance by inducing increased homozygosity and correlations within and across loci. AM likely leads to covariance of parental genetic effects, a measure of the similarity of parents in the indirect effects they have on their children. We propose that this assortment for parental characteristics, where biological parents create similar environments for their children, can create shared parental effects across traits and the appearance of cross-trait AM. Our theory shows how the resemblance among relatives increases under both AM, indirect and parent-of-origin effects. When our model is used to predict correlations among relatives in human height, we find that explaining the patterns observed in real data requires both indirect genetic effects and assortative mating. The degree to which direct, indirect and epigenetic effects shape the phenotypic variance of complex traits remains an open question that requires large-scale family data to be resolved.},
  author       = {Krätschmer, Ilse and Robinson, Matthew Richard},
  issn         = {1943-2631},
  journal      = {Genetics},
  number       = {4},
  publisher    = {Oxford University Press},
  title        = {{A quantitative genetic model for indirect genetic effects and genomic imprinting under random and assortative mating}},
  doi          = {10.1093/genetics/iyag042},
  volume       = {232},
  year         = {2026},
}

@phdthesis{22258,
  abstract     = {Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.

First, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. 

Second, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.

Collectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.},
  author       = {Depope, Al},
  issn         = {2663-337X},
  keywords     = {Approximate Message Passing, GWAS, Genomics, Proteomics, Survival modeling},
  pages        = {169},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics}},
  doi          = {10.15479/AT-ISTA-22258},
  year         = {2026},
}

@article{21488,
  abstract     = {Human height is a model for the genetic analysis of complex traits, and recent studies suggest the presence of thousands of common genetic variant associations and hundreds of low-frequency/rare variants. Here, we develop a new algorithmic paradigm based on approximate message passing (genomic vector approximate message passing [gVAMP]) for identifying DNA sequence variants associated with complex traits and common diseases in large-scale whole-genome sequencing (WGS) data. We show that gVAMP accurately localizes associations to variants with the correct frequency and position in the DNA, outperforming existing fine-mapping methods in selecting the appropriate genetic variants within WGS data. We then apply gVAMP to jointly model the relationship of tens of millions of WGS variants with human height in hundreds of thousands of UK Biobank individuals. We identify 59 rare variants and gene burden scores alongside many hundreds of DNA regions containing common variant associations and show that understanding the genetic basis of complex traits will require the joint analysis of hundreds of millions of variables measured on millions of people. The polygenic risk scores obtained from gVAMP have high accuracy (including a prediction accuracy of ∼46% for human height) and outperform current methods for downstream tasks such as mixed linear model association testing across 13 UK Biobank traits. In conclusion, gVAMP offers a scalable foundation for a wider range of analyses in WGS data.},
  author       = {Depope, Al and Bajzik, Jakub and Mondelli, Marco and Robinson, Matthew Richard},
  issn         = {2666-979X},
  journal      = {Cell Genomics},
  number       = {5},
  publisher    = {Elsevier},
  title        = {{Joint modeling of whole-genome sequencing data for human height via approximate message passing}},
  doi          = {10.1016/j.xgen.2026.101162},
  volume       = {6},
  year         = {2026},
}

@article{22614,
  abstract     = {This study presents a sustainable and non-destructive recycling strategy for synthetic fibres, specifically polyethylene terephthalate (PET), polyamide 6 (PA6), polyamide 66 (PA66), and elastane (EL). Two deep eutectic solvents (DESs) were synthesised, enabling selective dissolution of EL without degrading the surrounding polymer matrices. The dissolved EL phase was subsequently recovered by centrifugation, demonstrating its potential for material reuse. Comprehensive characterisation confirmed that the process preserves fibre integrity. SEM imaging revealed no detectable changes in fibre morphology, while DSC and TGA analyses indicated that the thermal properties of both EL and synthetic polymers were preserved. ATR-FTIR spectroscopy verified the absence of measurable chemical modifications. Mechanical testing showed that solvent treatment did not compromise the tensile performance of PET, PA6, or PA66 fibres, confirming their suitability for fibre-to-fibre recycling.
The recovered fibres were successfully re-extruded into continuous filaments, confirming their melt processability and enabling potential applications in textile yarns, technical fibres, nonwovens, and polymer components via melt-spinning or moulding. These results demonstrate that the proposed DES-based method offers a promising approach for the circular recycling of complex fibre blends and provides a viable pathway towards industrial implementation.},
  author       = {Depope, Nika and Depope, Al and Archodoulaki, Vasiliki Maria and Dabrowska, Alicja and Lendl, Bernhard and Mautner, Andreas and Ipsmiller, Wolfgang and Bartl, Andreas},
  issn         = {1879-2456},
  journal      = {Waste Management},
  publisher    = {Elsevier},
  title        = {{DES formulation for recycling synthetic fibre textile waste containing elastane}},
  doi          = {10.1016/j.wasman.2026.115736},
  volume       = {224},
  year         = {2026},
}

@article{21987,
  abstract     = {We introduce JODIE, a genetic joint modeling approach that estimates how DNA loci influence human traits by partitioning genetic effects into four components: direct effects (from a child’s alleles), indirect maternal and paternal effects (from parents’ alleles), and parent-of-origin (PofO) effects (dependent on parental transmission of alleles), while uniquely accounting for assortative mating. We analyze 30,000 child-mother-father trios from the Estonian Biobank and the Norwegian Mother, Father, and Child Cohort, focusing on height, body mass index, and childhood educational test scores. We find direct effects to be the largest contributor to trait variation, but combined, indirect parental and PofO effects are similarly substantial. We support our results by within-family genome-wide association testing and identify 276 independently associated DNA regions with a complex interplay between direct, indirect, and PofO effects. By joint modeling, we show that direct, indirect, and PofO effects collectively shape human phenotypic variation across loci genome-wide.},
  author       = {Krätschmer, Ilse and Hegemann, Laura and Hofmeister, Robin J. and Corfield, Elizabeth C. and Mahmoudi, Mahdi and Delaneau, Olivier and Andreassen, Ole A. and Campbell, Archie and Hayward, Caroline and Marioni, Riccardo E. and Ystrom, Eivind and Havdahl, Alexandra and Robinson, Matthew Richard},
  issn         = {2666-979X},
  journal      = {Cell Genomics},
  keywords     = {direct genetic effects, DGE, indirect genetic effects, IGE, parent-of-origin effects, phenotypic variation, assortative mating, within-family GWAS, MoBa, EstBB},
  number       = {7},
  publisher    = {Elsevier},
  title        = {{Separating direct, indirect, and parent-of-origin genetic effects in the human population}},
  doi          = {10.1016/j.xgen.2026.101277},
  volume       = {6},
  year         = {2026},
}

@article{18754,
  abstract     = {Exploring the molecular correlates of metabolic health measures may identify their shared and unique biological processes and pathways. Molecular proxies of these traits may also provide a more objective approach to their measurement. Here, DNA methylation (DNAm) data were used in epigenome-wide association studies (EWASs) and for training epigenetic scores (EpiScores) of six metabolic traits: body mass index (BMI), body fat percentage, waist-hip ratio, and blood-based measures of glucose, high-density lipoprotein cholesterol, and total cholesterol in >17,000 volunteers from the Generation Scotland (GS) cohort. We observed a maximum of 12,033 significant findings (p < 3.6 × 10−8) for BMI in a marginal linear regression EWAS. By contrast, a joint and conditional Bayesian penalized regression approach yielded 27 high-confidence associations with BMI. EpiScores trained in GS performed well in both Scottish and Singaporean test cohorts (Lothian Birth Cohort 1936 [LBC1936] and Health for Life in Singapore [HELIOS]). The EpiScores for BMI and total cholesterol performed best in HELIOS, explaining 20.8% and 7.1% of the variance in the measured traits, respectively. The corresponding results in LBC1936 were 14.4% and 3.2%, respectively. Differences were observed in HELIOS for body fat, where the EpiScore explained ∼9% of the variance in Chinese and Malay -subgroups but ∼3% in the Indian subgroup. The EpiScores also correlated with cognitive function in LBC1936 (standardized βrange: 0.08–0.12, false discovery rate p [pFDR] < 0.05). Accounting for the correlation structure across the methylome can vastly affect the number of lead findings in EWASs. The EpiScores of metabolic traits are broadly applicable across populations and can reflect differences in cognition.},
  author       = {Smith, Hannah M. and Ng, Hong Kiat and Moodie, Joanna E. and Gadd, Danni A. and Mccartney, Daniel L. and Bernabeu, Elena and Campbell, Archie and Redmond, Paul and Taylor, Adele and Page, Danielle and Corley, Janie and Harris, Sarah E. and Tay, Darwin and Deary, Ian J. and Evans, Kathryn L. and Robinson, Matthew Richard and Chambers, John C. and Loh, Marie and Cox, Simon R. and Marioni, Riccardo E. and Hillary, Robert F.},
  issn         = {1537-6605},
  journal      = {American Journal of Human Genetics},
  number       = {1},
  pages        = {106--115},
  publisher    = {Elsevier},
  title        = {{DNA methylation-based predictors of metabolic traits in Scottish and Singaporean cohorts}},
  doi          = {10.1016/j.ajhg.2024.11.012},
  volume       = {112},
  year         = {2025},
}

@article{19023,
  abstract     = {Alcohol consumption is an important risk factor for multiple diseases. It is typically assessed via self-report, which is open to measurement error through recall bias. Instead, molecular data such as blood-based DNA methylation (DNAm) could be used to derive a more objective measure of alcohol consumption by incorporating information from cytosine-phosphate-guanine (CpG) sites known to be linked to the trait. Here, we explore the epigenetic architecture of self-reported weekly units of alcohol consumption in the Generation Scotland study. We first create a blood-based epigenetic score (EpiScore) of alcohol consumption using elastic net penalized linear regression. We explore the effect of pre-filtering for CpG features ahead of elastic net, as well as differential patterns by sex and by units consumed in the last week relative to an average week. The final EpiScore was trained on 16,717 individuals and tested in four external cohorts: the Lothian Birth Cohorts (LBC) of 1921 and 1936, the Sister Study, and the Avon Longitudinal Study of Parents and Children (total N across studies > 10,000). The maximum Pearson correlation between the EpiScore and self-reported alcohol consumption within cohort ranged from 0.41 to 0.53. In LBC1936, higher EpiScore levels had significant associations with poorer global brain imaging metrics, whereas self-reported alcohol consumption did not. Finally, we identified two novel CpG loci via a Bayesian penalized regression epigenome-wide association study of alcohol consumption. Together, these findings show how DNAm can objectively characterize patterns of alcohol consumption that associate with brain health, unlike self-reported estimates.},
  author       = {Bernabeu, Elena and Chybowska, Aleksandra D. and Kresovich, Jacob K. and Suderman, Matthew and Mccartney, Daniel L. and Hillary, Robert F. and Corley, Janie and Valdés-Hernández, Maria Del C. and Maniega, Susana Muñoz and Bastin, Mark E. and Wardlaw, Joanna M. and Xu, Zongli and Sandler, Dale P. and Campbell, Archie and Harris, Sarah E. and Mcintosh, Andrew M. and Taylor, Jack A. and Yousefi, Paul and Cox, Simon R. and Evans, Kathryn L. and Robinson, Matthew Richard and Vallejos, Catalina A. and Marioni, Riccardo E.},
  issn         = {1868-7083},
  journal      = {Clinical Epigenetics},
  publisher    = {Springer Nature},
  title        = {{Blood-based epigenome-wide association study and prediction of alcohol consumption}},
  doi          = {10.1186/s13148-025-01818-y},
  volume       = {17},
  year         = {2025},
}

@article{20479,
  abstract     = {Genetic variation is generally regarded as a prerequisite for evolution. In principle, epigenetic information inherited independently of DNA sequence can also enable evolution, but whether this occurs in natural populations is unknown. Here we show that single-nucleotide and epigenetic gene body DNA methylation (gbM) polymorphisms explain comparable amounts of expression variance in <jats:italic>Arabidopsis thaliana</jats:italic> populations. We genetically demonstrate that gbM regulates transcription, and we identify and genetically validate many associations between gbM polymorphism and the variation of complex traits: fitness under heat and drought, flowering time and accumulation of diverse minerals. Epigenome-wide association studies pinpoint trait-relevant genes with greater precision than genetic association analyses, probably due to reduced linkage disequilibrium between gbM variants. Finally, we identify numerous associations between gbM epialleles and diverse environmental conditions in native habitats, suggesting that gbM facilitates adaptation. Overall, our results indicate that epigenetic methylation variation fundamentally shapes phenotypic diversity in a natural population.},
  author       = {Shahzad, Zaigham and Hollwey, Elizabeth and Moore, Jonathan D. and Choi, Jaemyung and Cassin-Ross, Gaëlle and Rouached, Hatem and Robinson, Matthew Richard and Zilberman, Daniel},
  issn         = {2055-0278},
  journal      = {Nature Plants},
  pages        = {2084--2099},
  publisher    = {Springer Nature},
  title        = {{Gene body methylation regulates gene expression and mediates phenotypic diversity in natural Arabidopsis populations}},
  doi          = {10.1038/s41477-025-02108-4},
  volume       = {11},
  year         = {2025},
}

@article{20491,
  abstract     = {Global fibre production has expanded rapidly, with polyester and cotton dominating, significantly contributing to textile waste and increasing demand for sustainable solutions. This study presents innovative method to recycle polyester/cotton (PET/CO) blends using hydrophobic deep eutectic solvents (DESs), eliminating the need for toxic chemicals while achieving high dissolution yields. PET was completely dissolved within 5 min, substantially outperforming state-of-the-art methods and facilitating the efficient and selective recovery of both components, PET (97%) and CO (100%). SEM imaging confirmed no morphological changes in cotton fibres after treatment. The thermal stability of the recovered materials was validated using DSC and TGA analyses, while ATR-FTIR spectroscopy indicated no chemical changes. Mechanical testing confirmed recovered cotton’s tenacity and elongation are within expected ranges despite showing a decrease of 28% in tenacity and 34% in elongation. Hence, the proposed process provides an efficient and sustainable recycling solution for PET/CO blends, retaining both polymers in a condition similar to virgin materials used in textile manufacturing with minimal processing time.},
  author       = {Depope, Nika and Depope, Al and Archodoulaki, Vasiliki Maria and Ipsmiller, Wolfgang and Bartl, Andreas},
  issn         = {1879-2456},
  journal      = {Waste Management},
  publisher    = {Elsevier},
  title        = {{Deep eutectic solvent as a solution for polyester/cotton textile recycling}},
  doi          = {10.1016/j.wasman.2025.115177},
  volume       = {208},
  year         = {2025},
}

@article{20816,
  abstract     = {Background: DNA methylation (DNAm) can regulate gene expression, and its genome-wide patterns (epigenetic scores or EpiScores) can act as biomarkers for complex traits. The relative stability of methylation profiles may enable better assessment of chronic exposures compared to single time-point protein measures. We present the first large-scale epigenetic study of the highly-abundant serum proteome measured via ultra-high throughput mass spectrometry in 14,671 samples from the Generation Scotland cohort. We further demonstrate the first large-scale comparison of protein EpiScores and their respective proteins as predictors of incident cardiovascular disease.

Results: Marginal epigenome-wide association models, adjusting for age, sex, measurement batch, estimated white cell proportions, BMI, smoking and methylation principal components, reveal 15,855 significant CpG – protein associations across 125 of 133 proteins PBonferroni < 2.71 × 10-10. Bayesian epigenome-wide association studies of the same 133 proteins reveal 697 CpG-Protein associations (posterior inclusion probability > 0.95). 112 protein EpiScores correlate significantly with their respective protein in a holdout test-set. Of these, sixteen associate significantly with incident all-cause cardiovascular disease (Nevents=191) compared to one measured protein.

Conclusions: We highlight a complex interplay between the blood-based methylome and proteome. Importantly, we show that protein EpiScores correlate with measured proteins and demonstrate that the, as-yet understudied, high-abundance proteome may yield clinically relevant biomarkers. The protein EpiScores demonstrate more significant associations with cardiovascular disease than directly measured proteins, suggesting their potential as clinical biomarkers for monitoring or predicting disease risk. We suggest that biomarker development could be enhanced by the consideration of protein EpiScores alongside measured proteins.},
  author       = {Robertson, Josephine A. and Bajzik, Jakub and Vernardis, Spyros and Chybowska, Aleksandra D. and Mccartney, Daniel L. and Grauslys, Arturas and Mur, Jure and Smith, Hannah M. and Campbell, Archie and Drake, Camilla and Grant, Hannah and Pearce, Jamie and Russ, Tom C. and Adkin, Poppy and White, Matthew and Brigden, Charles and Messner, Christoph B. and Porteous, David J. and Hayward, Caroline and Cox, Simon R. and Zelezniak, Aleksej and Ralser, Markus and Robinson, Matthew Richard and Marioni, Riccardo E.},
  issn         = {1474-760X},
  journal      = {Genome Biology},
  publisher    = {Springer Nature},
  title        = {{Methylome-wide association studies and epigenetic biomarker development for 133 mass spectrometry-assessed circulating proteins in 14,671 Generation Scotland participants}},
  doi          = {10.1186/s13059-025-03892-0},
  volume       = {26},
  year         = {2025},
}

@article{14932,
  abstract     = {The huge antlers of the extinct Irish elk have invited evolutionary speculation since Darwin. In the 1970s, Stephen Jay Gould presented the first extensive data on antler size in the Irish elk and combined these with comparative data from other deer to test the hypothesis that the gigantic antlers were the outcome of a positive allometry that constrained large-bodied deer to have proportionally even larger antlers. He concluded that the Irish elk had antlers as predicted for its size and interpreted this within his emerging framework of developmental constraints as an explanatory factor in evolution. Here we reanalyze antler allometry based on new morphometric data for 57 taxa of the family Cervidae. We also present a new phylogeny for the Cervidae, which we use for comparative analyses. In contrast to Gould, we find that the antlers of Irish elk were larger than predicted from the allometry within the true deer, Cervini, as analyzed by Gould, but follow the allometry across Cervidae as a whole. After dissecting the discrepancy, we reject the allometric-constraint hypothesis because, contrary to Gould, we find no similarity between static and evolutionary allometries, and because we document extensive non-allometric evolution of antler size across the Cervidae.},
  author       = {Tsuboi, Masahito and Kopperud, Bjørn Tore and Matschiner, Michael and Grabowski, Mark and Syrowatka, Chrsitine and Pélabon, Christophe and Hansen, Thomas F.},
  issn         = {1934-2845},
  journal      = {Evolutionary Biology},
  pages        = {149--165},
  publisher    = {Springer Nature},
  title        = {{Antler allometry, the Irish elk and Gould revisited}},
  doi          = {10.1007/s11692-023-09624-1},
  volume       = {51},
  year         = {2024},
}

@inproceedings{17147,
  abstract     = {Efficient utilization of large-scale biobank data is crucial for inferring the genetic basis of disease and predicting health outcomes from the DNA. Yet we lack efficient, accurate methods that scale to data where electronic health records are linked to whole genome sequence information. To address this issue, our paper develops a new algorithmic paradigm based on Approximate Message Passing (AMP), which is specifically tailored for genomic prediction and association testing. Our method yields comparable out-of-sample prediction accuracy to the state of the art on UK Biobank traits, whilst dramatically improving computational complexity, with a 8x-speed up in the run time. In addition, AMP theory provides a joint association testing framework, which outperforms the currently used REGENIE method, in roughly a third of the compute time. This first, truly large-scale application of the AMP framework lays the foundations for a far wider range of statistical analyses for hundreds of millions of variables measured on millions of people.},
  author       = {Depope, Al and Mondelli, Marco and Robinson, Matthew Richard},
  booktitle    = {2024 IEEE International Conference on Acoustics, Speech, and Signal Processing},
  isbn         = {9798350344851},
  issn         = {1520-6149},
  location     = {Seoul, Korea},
  pages        = {13151--13155},
  publisher    = {IEEE},
  title        = {{Inference of genetic effects via approximate message passing}},
  doi          = {10.1109/ICASSP48485.2024.10447198},
  year         = {2024},
}

@phdthesis{17368,
  abstract     = {Recent advancements in molecular diagnostic techniques have enabled the collection of
multiple types of omics data from patients, including genomics, epigenomics, proteomics,
and transcriptomics. However, we lack effective methods for integrating all these different
data types and combining them with clinical outcomes to study the molecular mechanisms
that govern pathological phenotypes. We present multi-omics BayesW, a penalized Bayesian
regression method that can handle general omics data for survival analysis of time-to-event
phenotypes. Our method can: (1) accommodate incomplete data by allowing censored
individuals, (2) use continuous time-to-event data to test associations of markers with a
phenotype and (3) estimate effects jointly while allowing for independent groups of biological
markers. Extensive simulations using planted signals on real data demonstrate that our model
accurately retrieves the true parameters of the model while controlling for false discoveries
and maintaining the expected prediction accuracy. We address data correlations by estimating
the effects jointly, even between omic groups, while also estimating the individual variance
explained by each group. We apply our model to two datasets. Using 18,000 individuals from
the Generation Scotland study we model the association of time at onset of Type 2 Diabetes,
Stroke, Ischemic Disease, and Osteoarthritis from baseline study entry, with 831,724 CpG
methylation probes. We find that large proportions of variation in disease onset times can
be attributed to methylation as measured in whole blood at baseline in individuals without
disease symptoms. We then apply our model to The Cancer Genome Atlas (TCGA) pan-cancer
dataset, in which we use 5 types of omics: copy number variation, epigenetics, somatic
mutations, miRNA, and gene expression. For cancer survival age-at-onset we find that, when
fitting the 5 groups together, almost all variation attributable to "omics" data is explained by
DNA methylation. When considering progression times, both methylation and gene expression
explain a large part of the variance. We found 2 genes that are significantly associated (95%
posterior inclusion probability) with cancer survival time, conditional on all other genome-wide
omics data variation. Owing to the vast variability of mechanisms characterizing different
cancers, there are likely few specific genes with a strong signal in a pan-cancer setting. Taken
together, we showed the applicability of our multi-omics BayesW model to a wide-range of
biological questions in multi-omics data.
},
  author       = {Villanueva Marijuan, Ariadna},
  issn         = {2791-4585},
  keywords     = {Epigenetics, Multi-omics, Bayesian regression},
  pages        = {60},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Bayesian linear regression for analyzing general omics data with time-to-event phenotypes}},
  doi          = {10.15479/at:ista:17368},
  year         = {2024},
}

@phdthesis{18642,
  abstract     = {This thesis consists of two pieces of work in the broader field of computational biology,
both of which are methods for the analysis of large scale biological data, implemented in
efficient software.
Chapter 2 introduces a statistical software for causal discovery and inference from observed
genetic marker and phenotypic trait data. We explore in simulation how well the method
can fine-map genetic effects, find the correct causal structure among tens of traits and
millions of genetic markers, and infer the causal effect size for the discovered causal
relations. We then apply the method to 8 million markers and 17 traits from the UK
Biobank and show that many relationships found with other methods are likely due to
the effects of hidden confounders.
Chapter 3 describes how this method can be applied to longitudinal data. I show how one
can incorporate the background knowledge present in the known order of measurements to
improve the accuracy of the causal discovery process, and explore the method’s ability to
identify age specific genetic effects, and how the error rates of this recovery are influenced
by missing data due to different censoring mechanisms.
Chapter 4 introduces a statistical software for the comparison of chromatin contact maps
based on the structural similarity index. We explore the robustness of the method to
noise and size differences of the compared maps, show how it can measure evolutionary
conservation of topological features by providing a similarity ranking of syntenic regions,
and finally how it can detect alterations in 3D genome structure due to genetic mutations
in samples of medical relevance.
},
  author       = {Machnik, Nick N},
  issn         = {2663-337X},
  pages        = {138},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Algorithms for causal learning and comparative analysis for genomic data}},
  doi          = {10.15479/at:ista:18642},
  year         = {2024},
}

@unpublished{18648,
  abstract     = {Statistical causal learning in genomics relies on the instrumental variable method of
Mendelian Randomization (MR). Currently, an overwhelming number of MR studies
purport to show causal relationships among a wide range of risk factors and outcomes.
Here, we show that selecting instrument variables from genome-wide association study
estimates leads to high false discovery rates for many MR approaches, which can be
greatly reduced by employing a graphical inference approach which: (i) explicitly tests
instrumental variable assumptions; (ii) distinguishes direct from indirect factors in very
high-dimensional data; (iii) discriminates pleiotropic from trait-specific markers, controlling for LD genome-wide; (iv) accommodates rare variants and binary outcomes in a
principled way; and (v) identifies potential unobserved latent confounding. For 17 traits
and 8.4M variants recorded for 458,747 individuals in the UK Biobank, we show that
standard MR analysis gives an abundance of findings that disappear under stringent
assumption checks, with many relationships reflecting potential unmeasured confounding. This implies that mixtures of temporal precedence and potential for reverse-causality
prohibit understanding the underlying nature of phenotypic and genetic correlations in
biobank data. We propose that well-curated longitudinal records are likely needed and
that our approach provides a first-step toward robust principled screening for potential
causal links.
},
  author       = {Machnik, Nick N and Mahmoudi, Seyed Mahdi and Borczyk, Malgorzata and Krätschmer, Ilse and Bauer, Markus J. and Robinson, Matthew Richard},
  booktitle    = {bioRxiv},
  title        = {{Causal inference for multiple risk factors and diseases from genomics data}},
  doi          = {10.1101/2023.12.06.570392},
  year         = {2024},
}

@article{12719,
  abstract     = {Background
Epigenetic clocks can track both chronological age (cAge) and biological age (bAge). The latter is typically defined by physiological biomarkers and risk of adverse health outcomes, including all-cause mortality. As cohort sample sizes increase, estimates of cAge and bAge become more precise. Here, we aim to develop accurate epigenetic predictors of cAge and bAge, whilst improving our understanding of their epigenomic architecture.

Methods
First, we perform large-scale (N = 18,413) epigenome-wide association studies (EWAS) of chronological age and all-cause mortality. Next, to create a cAge predictor, we use methylation data from 24,674 participants from the Generation Scotland study, the Lothian Birth Cohorts (LBC) of 1921 and 1936, and 8 other cohorts with publicly available data. In addition, we train a predictor of time to all-cause mortality as a proxy for bAge using the Generation Scotland cohort (1214 observed deaths). For this purpose, we use epigenetic surrogates (EpiScores) for 109 plasma proteins and the 8 component parts of GrimAge, one of the current best epigenetic predictors of survival. We test this bAge predictor in four external cohorts (LBC1921, LBC1936, the Framingham Heart Study and the Women’s Health Initiative study).

Results
Through the inclusion of linear and non-linear age-CpG associations from the EWAS, feature pre-selection in advance of elastic net regression, and a leave-one-cohort-out (LOCO) cross-validation framework, we obtain cAge prediction with a median absolute error equal to 2.3 years. Our bAge predictor was found to slightly outperform GrimAge in terms of the strength of its association to survival (HRGrimAge = 1.47 [1.40, 1.54] with p = 1.08 × 10−52, and HRbAge = 1.52 [1.44, 1.59] with p = 2.20 × 10−60). Finally, we introduce MethylBrowsR, an online tool to visualise epigenome-wide CpG-age associations.

Conclusions
The integration of multiple large datasets, EpiScores, non-linear DNAm effects, and new approaches to feature selection has facilitated improvements to the blood-based epigenetic prediction of biological and chronological age.},
  author       = {Bernabeu, Elena and Mccartney, Daniel L. and Gadd, Danni A. and Hillary, Robert F. and Lu, Ake T. and Murphy, Lee and Wrobel, Nicola and Campbell, Archie and Harris, Sarah E. and Liewald, David and Hayward, Caroline and Sudlow, Cathie and Cox, Simon R. and Evans, Kathryn L. and Horvath, Steve and Mcintosh, Andrew M. and Robinson, Matthew Richard and Vallejos, Catalina A. and Marioni, Riccardo E.},
  issn         = {1756-994X},
  journal      = {Genome Medicine},
  publisher    = {Springer Nature},
  title        = {{Refining epigenetic prediction of chronological and biological age}},
  doi          = {10.1186/s13073-023-01161-y},
  volume       = {15},
  year         = {2023},
}

@article{12758,
  abstract     = {AlphaFold changed the field of structural biology by achieving three-dimensional (3D) structure prediction from protein sequence at experimental quality. The astounding success even led to claims that the protein folding problem is “solved”. However, protein folding problem is more than just structure prediction from sequence. Presently, it is unknown if the AlphaFold-triggered revolution could help to solve other problems related to protein folding. Here we assay the ability of AlphaFold to predict the impact of single mutations on protein stability (ΔΔG) and function. To study the question we extracted the pLDDT and <pLDDT> metrics from AlphaFold predictions before and after single mutation in a protein and correlated the predicted change with the experimentally known ΔΔG values. Additionally, we correlated the same AlphaFold pLDDT metrics with the impact of a single mutation on structure using a large scale dataset of single mutations in GFP with the experimentally assayed levels of fluorescence. We found a very weak or no correlation between AlphaFold output metrics and change of protein stability or fluorescence. Our results imply that AlphaFold may not be immediately applied to other problems or applications in protein folding.},
  author       = {Pak, Marina A. and Markhieva, Karina A. and Novikova, Mariia S. and Petrov, Dmitry S. and Vorobyev, Ilya S. and Maksimova, Ekaterina and Kondrashov, Fyodor and Ivankov, Dmitry N.},
  issn         = {1932-6203},
  journal      = {PLoS ONE},
  number       = {3},
  publisher    = {Public Library of Science},
  title        = {{Using AlphaFold to predict the impact of single mutations on protein stability and function}},
  doi          = {10.1371/journal.pone.0282689},
  volume       = {18},
  year         = {2023},
}

@article{14258,
  abstract     = {There is currently little evidence that the genetic basis of human phenotype varies significantly across the lifespan. However, time-to-event phenotypes are understudied and can be thought of as reflecting an underlying hazard, which is unlikely to be constant through life when values take a broad range. Here, we find that 74% of 245 genome-wide significant genetic associations with age at natural menopause (ANM) in the UK Biobank show a form of age-specific effect. Nineteen of these replicated discoveries are identified only by our modeling framework, which determines the time dependency of DNA-variant age-at-onset associations without a significant multiple-testing burden. Across the range of early to late menopause, we find evidence for significantly different underlying biological pathways, changes in the signs of genetic correlations of ANM to health indicators and outcomes, and differences in inferred causal relationships. We find that DNA damage response processes only act to shape ovarian reserve and depletion for women of early ANM. Genetically mediated delays in ANM were associated with increased relative risk of breast cancer and leiomyoma at all ages and with high cholesterol and heart failure for late-ANM women. These findings suggest that a better understanding of the age dependency of genetic risk factor relationships among health indicators and outcomes is achievable through appropriate statistical modeling of large-scale biobank data.},
  author       = {Ojavee, Sven E. and Darrous, Liza and Patxot, Marion and Läll, Kristi and Fischer, Krista and Mägi, Reedik and Kutalik, Zoltan and Robinson, Matthew Richard},
  issn         = {1537-6605},
  journal      = {American Journal of Human Genetics},
  number       = {9},
  pages        = {1549--1563},
  publisher    = {Elsevier},
  title        = {{Genetic insights into the age-specific biological mechanisms governing human ovarian aging}},
  doi          = {10.1016/j.ajhg.2023.07.006},
  volume       = {110},
  year         = {2023},
}

@article{14689,
  author       = {Ing-Simmons, Elizabeth and Machnik, Nick N and Vaquerizas, Juan M.},
  issn         = {1546-1718},
  journal      = {Nature Genetics},
  number       = {12},
  pages        = {2053--2055},
  publisher    = {Springer Nature},
  title        = {{Reply to: Revisiting the use of structural similarity index in Hi-C}},
  doi          = {10.1038/s41588-023-01595-5},
  volume       = {55},
  year         = {2023},
}

