@phdthesis{21854,
  abstract     = {As neural-network-based models grow both in size and popularity, interest has grown in making the models smaller and more efficient to train. To that end, many methods have been proposed to prune models by reducing their number of nonzero parameters. Additionally, parameter-efficient fine-tuning, in which a much smaller number of parameters than the total contained in the model is updated during training, has become very popular, especially in the space of Large Language Models. At the same time, the increasingly routine deployment of machine learning in real-world applications has spurred a drive to make them more trustworthy - in the sense of, among other things, being unbiased, interpretable, and editable. In this thesis, we examine the interplay between efficiency and trustworthiness.

First, we analyze the effects of model pruning on bias in computer vision models, demonstrating that increased sparsity leads to greater bias, largely as a function of increased model uncertainty in marginal cases. Based on this observation, we propose several bias mitigation techniques. Then, we demonstrate that example-specific model pruning can improve model interpretation methods while improving pruning efficiency to make example-specific model pruning feasible in real time. Then, we investigate the effectiveness of parameter-efficient and data-efficient model personalization via fine-tuning, demonstrating that it is highly feasible with very small computational and data resources. Finally, we consider efficiency in editing model knowledge using a custom synthetic data framework, demonstrating that parameter-efficient, low-rank fine-tuning frequently outperforms full-rank fine-tuning, and, additionally, that restricting which model blocks are fine-tuned frequently improves results. Together, the results in this thesis provide new insights and techniques for combining trustworthiness and efficiency during neural network inference and training.

},
  author       = {Iofinova, Eugenia B},
  issn         = {2663-337X},
  pages        = {237},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{On the utility and effects of efficiency in artificial neural networks}},
  doi          = {10.15479/AT-ISTA-21854},
  year         = {2026},
}

@misc{21857,
  abstract     = {The availability of powerful open-source large language models (LLMs) opens exciting use cases, such as using personal data to fine-tune these models to imitate a user’s unique writing style. Two key requirements for this functionality are personalization–in the sense that the output should recognizably reflect the user’s own writing style—and privacy–users may justifiably be wary of uploading extremely personal data, such as their email archive, to a third-party service. In this paper, we demonstrate the feasibility of training and running such an assistant, which we call Panza, on commodity hardware, for the specific use case of email generation. Panza’s personalization features are based on a combination of parameter-efficient fine-tuning using a variant of the Reverse Instructions technique [1] and Retrieval-Augmented Generation (RAG) [2]. We demonstrate that this combination allows us to fine-tune an LLM to reflect a user’s writing style using limited data, while executing on extremely limited resources, e.g. on a free Google Colab instance. Our key methodological contribution is the first detailed study of evaluation metrics for this task, and
of how different choices of system components–the use of RAG and of different fine-tuning approaches–impact the system’s performance. Additionally, we demonstrate that very little data - under 100 email samples - are sufficient to create models that convincingly imitate humans, showcasing a previously unknown attack vector in language models. We are releasing the full Panza code as well as three new email datasets licensed for research use.},
  author       = {Nicolicioiu, Armand and Iofinova, Eugenia B and Jovanovic, Andrej and Kurtic, Eldar and Nikdan, Mahdi and Panferov, Andrei and Markov, Ilia and Shavit, Nir and Alistarh, Dan-Adrian},
  booktitle    = {Third Conference on Parsimony and Learning (Proceedings Track)},
  keywords     = {LLMs, PEFT, LoRA, personalization, efficient ML},
  location     = {Tübíngen, Germany},
  publisher    = {OpenReview},
  title        = {{Panza: Investigating the feasibility of fully-local personalized text generation}},
  year         = {2026},
}

@unpublished{21859,
  abstract     = {As artificial neural networks, and specifically large language models, have improved rapidly in capabilities and quality, they have increasingly been deployed in real-world applications, from customer service to Google search, despite the fact that they frequently make factually incorrect or undesirable statements. This trend has inspired practical and academic interest in model editing, that is, in adjusting the weights of the model to modify its likely outputs for queries relating to a specific fact or set of facts. This may be done either to amend a fact or set of facts, for instance, to fix a frequent error in the training data, or to suppress a fact or set of facts entirely, for instance, in case of dangerous knowledge. Multiple methods have been proposed to do such edits. However, at the same time, it has been shown that such model editing can be brittle and incomplete. Moreover the effectiveness of any model editing method necessarily depends on the data on which the model is trained, and, therefore, a good understanding of the interaction of the training data distribution and the way it is stored in the network is necessary and helpful to reliably perform model editing. However, working with large language models trained on real-world data does not allow us to understand this relationship or fully measure the effects of model editing. We therefore propose Behemoth, a fully synthetic data generation framework. To demonstrate the practical insights from the framework, we explore model editing in the context of simple tabular data, demonstrating surprising findings that, in some cases, echo real-world results, for instance, that in some cases restricting the update rank results in a more effective update.},
  author       = {Iofinova, Eugenia B and Alistarh, Dan-Adrian},
  booktitle    = {arXiv},
  title        = {{Behemoth: Benchmarking unlearning in LLMs using fully synthetic data}},
  doi          = {10.48550/arXiv.2601.23153},
  year         = {2026},
}

@misc{21422,
  author       = {Sunko, Veronika},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Data underpinning "Magneto-optical Kerr effect in an A-type antiferromagnet"}},
  doi          = {10.15479/AT-ISTA-21422},
  year         = {2026},
}

@article{21759,
  abstract     = {Promoters and enhancers are cis-regulatory elements (CREs), DNA sequences that bind transcription factor (TF) proteins to up- or down-regulate target genes. Decades-long efforts yielded TF-DNA interaction models that predict how strongly an individual TF binds arbitrary DNA sequences and how individual binding events on the CRE combine to affect gene expression. These insights can be synthesized into a global, biophysically realistic, and quantitative genotype-phenotype (GP) map for gene regulation, a ‘holy grail’ for the application of evolutionary theory. A global map provides a rare opportunity to simulate the long-term evolution of regulatory sequences and pose several fundamental questions: How long does it take to evolve CREs de novo? How many non-trivial regulatory functions exist in sequence space? How connected are they? For which regulatory architecture is CRE evolution most rapid and evolvable? In this article, the second of a two-part series, we review the application of evolutionary concepts — epistasis, robustness, evolvability, tunability, plasticity, and bet-hedging — to the evolution of gene regulatory sequences. We then evaluate the potential for a unifying theory for the evolution of regulatory sequences and identify key open challenges.},
  author       = {Mascolo, Elia and Körei, Reka E and Borst, Noa O. and Barton, Nicholas H and Crocker, Justin and Tkačik, Gašper},
  issn         = {1879-0380},
  journal      = {Current Opinion in Genetics and Development},
  publisher    = {Elsevier},
  title        = {{Long-term evolution of regulatory DNA sequences. Part 2: Theory and future challenges}},
  doi          = {10.1016/j.gde.2026.102472},
  volume       = {98},
  year         = {2026},
}

@article{21849,
  abstract     = {The development of complex tissues relies on the precise assignment of cell identity. At the molecular scale, this process depends on the deposition of epigenetic modifications—such as methylation—that are regulated by complex biochemical networks and occur at specific regions on the DNA and chromatin. Here we show that despite the complexity of epigenetic regulation, dynamical scaling and self-similarity of DNA methylation marks emerge in embryonic development. Drawing on single-cell multi-omics experiments, super-resolution microscopy and statistical physics, we demonstrate that these phenomena originate in dynamical feedback between DNA methylation and the formation of nanoscale dynamic chromatin aggregates. These nanoscale processes lead to genome-wide increase in DNA methylation marks following a power law and self-similar correlation functions. Using this framework, we identify methylation patterns that precede gene expression changes in embryonic symmetry breaking. Our work identifies linear sequencing measurements as a laboratory to study mesoscopic biophysical processes in vivo.},
  author       = {Olmeda, Fabrizio and Lohoff, Tim and Kafetzopoulos, Ioannis and Clark, Stephen J. and Benson, Laura and Santos, Fatima and Krueger, Felix and Walker, Simon and Reik, Wolf and Rulands, Steffen},
  issn         = {1745-2481},
  journal      = {Nature Physics},
  pages        = {931--940},
  publisher    = {Springer Nature},
  title        = {{Scaling and self-similarity in the formation of the embryonic epigenome}},
  doi          = {10.1038/s41567-026-03263-x},
  volume       = {22},
  year         = {2026},
}

@article{21883,
  abstract     = {Three-dimensional (3D) printing has rapidly developed from a niche hobbyist activity into a widely accessible and indispensable technology across multiple scientific disciplines. Within microscopy, optical engineering laboratories and imaging core facilities, 3D printing enables creating customised solutions for sample holders, optical components and everyday laboratory tools that traditionally required specialised machining. By providing rapid prototyping, low-cost production and reproducibility, 3D printing facilitates innovation and efficiency in facility operations. This article provides a perspective on the possibilities, challenges, and practical aspects of implementing 3D printing within microscopy core facilities. Instead of providing technical review about 3D printing, we focus on service organisation, user engagement, resource management and community-driven repositories for design dissemination. Our aim is to share insights with those considering the implementation of 3D printing as a service for developing add-on components to ease the operation of different aspects of the machine-park driven services and those who are managing advanced instrumentation within research groups.},
  author       = {Goudarzi, Mohammad and Schuster, Maximilian and Milberger, Arthur and Gunkel, Manuel and Terjung, Stefan and Krens, Gabriel},
  issn         = {1365-2818},
  journal      = {Journal of Microscopy},
  number       = {3},
  pages        = {382--395},
  publisher    = {Wiley},
  title        = {{3D printing in core facilities – Low pain, high gain}},
  doi          = {10.1111/jmi.70106},
  volume       = {302},
  year         = {2026},
}

@article{21872,
  abstract     = {Magneto-optic Kerr effect (MOKE) is a powerful probe of broken time-reversal symmetry (T), typically used to study ferromagnets. While MOKE has been observed in some antiferromagnets (AFMs) with vanishing magnetization, it is often associated with structures whose symmetry is lower than basic collinear, bipartite order. In contrast, theory predicts a mechanism for MOKE intrinsic to all AFMs of A-type, i.e. layered AFMs in which ferromagnetic layers are antiferromagnetically aligned. Here we report the experimental confirmation of this mechanism in a bulk AFM. We achieve this by measuring the imaginary component of MOKE as a function of photon energy in MnBi2Te4, an A-type AFM where T is preserved in combination with a translation, and comparing the experimental results with model calculations. Our model suggests that observable MOKE should be expected in all collinear A-type AFMs with out-of-plane spin order, thus enabling optical detection of AFM domains and expanding the scope of MOKE to few-layer AFMs.},
  author       = {Sunko, Veronika and Ahsanullah, Salman and Jain, Vivek and Weber, Sophie and Kumaran, Sivaloganathan and Yan, Jiaqiang and Orenstein, Joseph and Ovchinnikov, Dmitry},
  issn         = {2041-1723},
  journal      = {Nature Communications},
  publisher    = {Springer Nature},
  title        = {{Magneto-optical Kerr effect in an A-type antiferromagnet}},
  doi          = {10.1038/s41467-026-72577-4},
  volume       = {17},
  year         = {2026},
}

@article{21950,
  abstract     = {One Health initiatives are modern paradigms for research and health care practices in various fields. Concrete definitions of the One Health framework, however, remain heterogeneous, leading to conceptual problems and uncertainties in the application of the framework. This article discusses several approaches to the One Health concept, and their associated consequences, with special focus on animal experimentation. The first issue addressed is how One Health should be defined, as well as what (and who) should be considered within a One Health approach. In order to shed further light on this, we explore the history of animals in biomedical science, highlighting historical milestones in the use of animal models, as well as the development and current state of ethical considerations in the field of animal experimentation. The second issue comes with the inclusion of animal experimentation per se as part of the One Health concept. Therefore, particular attention is paid to bioethical principles and the resulting problems that can arise when applying them to the One Health concept. Arguments such as the idea of inequality between humans and non-human animals, and the premise that all actions are done for the benefit of humans, are raised and then used to explore the question of whether the One Health concept is compatible with existing bioethical principles. Based on the bioethical principles of protecting the environment, the biodiversity and biosphere, this paper seeks an inclusive perspective of the One Health concept. Successful solutions will be based on this concept, which embraces all living beings. The authors conclude that a multispecies ethics approach could help create a more ethical ecosystem that is aligned with the wellbeing of all life on a shared planet.},
  author       = {Ulman, Yesim Isil and Kostomitsopoulos, Nikos and Camenzind, Samuel and Kitsara, Maria and Pavone, Ilja Richard and Schober, Sophie},
  issn         = {2632-3559},
  journal      = {Alternatives to Laboratory Animals},
  number       = {4},
  pages        = {226--235},
  publisher    = {SAGE Publications},
  title        = {{Emerging bioethical conflicts: One Health and animal experimentation}},
  doi          = {10.1177/02611929261453330},
  volume       = {54},
  year         = {2026},
}

@article{21900,
  abstract     = {Individually silencing 125 fruit fly genes reveals opposing fitness effects of mutations between females and males, as well as between germline and somatic tissues.},
  author       = {Ruzicka, Filip},
  issn         = {2397-334X},
  journal      = {Nature Ecology & Evolution},
  pages        = {1035--1036},
  publisher    = {Springer Nature},
  title        = {{Reverse genetics of sexual antagonism}},
  doi          = {10.1038/s41559-026-03036-y},
  volume       = {10},
  year         = {2026},
}

@phdthesis{21957,
  abstract     = {This thesis investigates algorithmic certification and approximation methods for degenerate semidefinite programs (SDPs) and the singular roots of polynomial systems. In the first part, we present a hybrid symbolic-numeric algorithm for certifying the feasibility of weakly feasible, degenerate SDPs. By reformulating linear matrix inequalities (LMIs) into a structured polynomial system via facial reduction and incidence varieties, we guarantee the existence of an isolated exact solution. This algebraic reduction enables the certification of maximum-rank numerical approximations using methods from algebraic geometry.

In the second part, we address the severe ill-conditioning and loss of quadratic convergence that plague standard path-tracking methods near isolated singular roots. To overcome this, we propose tracking algorithms that achieve superlinear convergence without the computational bloat characteristic of classical deflation techniques. By modeling the solution path as a generalized fractional Puiseux series, our approach combines an explicitly derived algebraic predictor with a localized hyperplane desingularization phase during the corrector step. Furthermore, we introduce a continuous path-limit method and an extension of the geometric sequence rule to directly extract exact fractional exponents. This bypasses traditional heuristic trial-and-error methods and explicitly accommodates sparse series expansions. Numerical experiments confirm that our method significantly reduces the cumulative number of matrix inversions while achieving high-accuracy root approximations, even for heavily degenerate systems exhibiting higher coranks.},
  author       = {Zapata, Jeferson},
  isbn         = {978-3-99078-079-4},
  issn         = {2663-337X},
  pages        = {89},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Overcoming degeneracy and singularity: Techniques for semidefinite programs and homotopy continuation endgames}},
  doi          = {10.15479/AT-ISTA-21957},
  year         = {2026},
}

@phdthesis{21360,
  author       = {Riegler, Stefan},
  issn         = {2663-337X},
  pages        = {185},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Root system plasticity under nutrient limitation: Investigating hormonal and molecular drivers in Arabidopsis thaliana and Coffea  species}},
  doi          = {10.15479/AT-ISTA-21360},
  year         = {2026},
}

@misc{21363,
  abstract     = {The data contains information on coffee differential gene expression as well as co-expression and trait correlations in two separate experiments. First, contrasting nitrogen supply, second, intra- and interspecific grafting.},
  author       = {Riegler, Stefan},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Thesis Data for Root System Plasticity under Nutrient Limitation: Investigating Hormonal and Molecular Drivers in Arabidopsis thaliana and Coffea  species}},
  doi          = {10.15479/AT-ISTA-21363},
  year         = {2026},
}

@article{21164,
  abstract     = {Global emission inventories often fail to capture the complexities of vehicular pollution in regions with unique fuel mixes, such as Brazil’s extensive biofuel use, leading to significant uncertainties in atmospheric modeling. This study presents a century-long (1960–2100) bottom-up vehicular emission inventory for Brazil, leveraging locally derived emission factors. Our estimates reveal substantial discrepancies in magnitude, timing, and speciation of non-CO2 pollutants (CO, NMHC, PM2.5) compared to leading global inventories (EDGAR, CEDS, CAMS), highlighting critical inaccuracies in widely used data sets. More critically, future projections under Shared Socioeconomic Pathways (SSPs) uncover a novel positive feedback mechanism: rising temperatures significantly enhance vehicular evaporative nonmethane hydrocarbon (NMHC) emissions. This temperature-dependent increase and subsequent NMHC oxidation to CO2 suggest an overlooked pathway that could amplify climate warming and air pollution globally, particularly after a breakpoint around 2050 (p < 0.05). While historical emissions peaked in the 1990s–2000s, nonexhaust PM becomes increasingly important. Air quality simulations using our inventory in the MUSICA model show good regional PM2.5 agreement but highlight challenges in resolving local primary pollutant peaks. This comprehensive inventory provides crucial data for Brazil and uncovers globally relevant climate–chemistry interactions, urging a re-evaluation of regional specificities in global emission assessments.},
  author       = {Ibarra-Espinosa, Sergio and Dias de Freitas, Edmilson and Gaubert, Benjamin and Lichtig, Pablo and Ropkins, Karl and da Silva, Iara and Martins Pereira, Guilherme and Schuch, Daniel and Nascimento, Janaina and Hoinaski, Leonardo and Martins, Leila Droprinchinski and Gavidia-Calderón, Mario and Vara-Vela, Angel and Toledo de Almeida Albuquerque, Taciana and Ynoue, Rita Yuri and Diez, Sebastian and Mera, Zamir and Casallas Garcia, Alejandro and Vallejo, Fidel and Diaz, Valeria and Pedruzzi, Rizzieri and Abrutzky, Rosana and Franco, Marco A. and Huneeus, Nicolas and Jorquera, Hector and Belalcázar-Cerón, Luis Carlos and Rojas, Néstor Y. and de Fatima Andrade, Maria and Emmons, Louisa and Brasseur, Guy},
  issn         = {1520-5851},
  journal      = {Environmental Science &amp; Technology},
  number       = {6},
  publisher    = {American Chemical Society},
  title        = {{A century of vehicular emissions in Brazil: Unveiling the impacts of unique fuel mix on air quality}},
  doi          = {10.1021/acs.est.5c08400},
  volume       = {60},
  year         = {2026},
}

@article{20971,
  abstract     = {Mountain glaciers are among the natural systems most vulnerable to climate change. However, their interactions with the atmosphere are complex and not fully understood. These interactions can trigger rapid adjustments and climate feedbacks that either amplify or attenuate atmospheric signals, influencing both glacier response and large-scale atmospheric circulation. Observing this functional coupling in nature is challenging because the key processes occur over a wide range of spatial and temporal scales. However, recent advances in observational techniques and modeling have provided new insights into these interactions. In this review, we summarize the current state of knowledge on glacier-atmosphere interactions in high-mountain regions at different scales, and highlight recent advances in observational and numerical modeling. We also highlight important knowledge gaps and outline future research directions to improve the prediction of glacier change in a warming world.},
  author       = {Sauter, T. and Brock, B. W. and Collier, E. and Goger, B. and Groos, A. R. and Haualand, K. F. and Mott, R. and Nicholson, L. and Prinz, R. and Shaw, Thomas and Stiperski, I. and Georgi, A. and Haugeneder, M. and Mandal, A. and Reynolds, D. and Saigger, M. and Sicart, J. E. and Voordendag, A.},
  issn         = {1944-9208},
  journal      = {Reviews of Geophysics},
  number       = {1},
  publisher    = {Wiley},
  title        = {{Glacier-atmosphere interactions and feedbacks in high-mountain regions - A review}},
  doi          = {10.1029/2024RG000869},
  volume       = {64},
  year         = {2026},
}

@article{20935,
  abstract     = {In situ cryo-electron tomography (cryo-ET) has emerged as the method of choice to investigate the structures of biomolecules in their native context. However, challenges remain for the efficient production and sharing of large-scale cryo-ET datasets. Here, we combined cryogenic plasma-based focused ion beam (cryo-PFIB) milling with recent advances in cryo-ET acquisition and processing to generate a dataset of 1,829 annotated tomograms of the green alga Chlamydomonas reinhardtii, which we provide as a community resource to drive method development and inspire biological discovery. To assay data quality, we performed subtomogram averaging of both soluble and membrane-bound complexes ranging in size from >3 MDa to ∼200 kDa, including 80S ribosomes, Rubisco, nucleosomes, microtubules, clathrin, photosystem II, and mitochondrial ATP synthase. The majority of these density maps reached sub-nanometer resolution, demonstrating the potential of this C. reinhardtii dataset as well as the promise of modern cryo-ET workflows and open data sharing to empower visual proteomics.},
  author       = {Kelley, Ron and Khavnekar, Sagar and Righetto, Ricardo D. and Heebner, Jessica and Obr, Martin and Zhang, Xianjun and Chakraborty, Saikat and Tagiltsev, Grigory and Michael, Alicia and Van Dorst, Sofie and Waltz, Florent and Mccafferty, Caitlyn L. and Lamm, Lorenz and Zufferey, Simon and Van Der Stappen, Philippe and Van Den Hoek, Hugo and Wietrzynski, Wojciech and Harar, Pavol and Wan, William and Briggs, John A.G. and Plitzko, Jürgen M. and Engel, Benjamin D. and Kotecha, Abhay},
  issn         = {1097-4164},
  journal      = {Molecular Cell},
  number       = {1},
  pages        = {213--230.e7},
  publisher    = {Elsevier},
  title        = {{Toward community-driven visual proteomics with large-scale cryo-electron tomography of Chlamydomonas reinhardtii}},
  doi          = {10.1016/j.molcel.2025.11.029},
  volume       = {86},
  year         = {2026},
}

@article{20858,
  abstract     = {Targeted antigen delivery to immune cells, particularly dendritic cells, has emerged as a promising strategy to enhance therapeutic efficacy of vaccines, while minimizing adverse effects associated with conventional immunization. In this study, we use our previously described small glycomimetic molecule that is selectively recognized by the Langerhans cell (LC)-specific surface receptor Langerin and demonstrate specific delivery of protein antigens to these specialized dendritic cells. Our results show that Langerin-mediated antigen delivery significantly enhances the immune response in vivo, resulting in increased expansion and activation of antigen-specific T cells, compared to immunization with unmodified antigen. We demonstrate the feasibility of our LC-targeted platform for immune cell-specific immunization with protein antigen and underscore the potential of LCs as an access point for next-generation vaccines and immunotherapies.},
  author       = {Rica, Ramona and Klein, Klara and Johnson, Litty and Carta, Gabriele and Sarcevic, Mirza and Langer, Freyja and Rademacher, Christoph and Wawrzinek, Robert and Quattrone, Federica and Sparber, Florian},
  issn         = {1525-0024},
  journal      = {Molecular Therapy},
  number       = {1},
  pages        = {397--406},
  publisher    = {Elsevier},
  title        = {{Langerhans cell-targeted protein delivery enhances antigen-specific cellular immune response}},
  doi          = {10.1016/j.ymthe.2025.10.008},
  volume       = {34},
  year         = {2026},
}

@article{20537,
  abstract     = {In this personal account, I describe the work performed in my research group on the development of methods that harness heterogeneous photocatalysts for light-mediated nickel-catalyzed cross-couplings. This includes catalytic systems using carbon nitride materials, dye-sensitized TiO₂, covalent organic frameworks (COFs), and conjugated polymers. The rationale behind the selection of materials and how their use led to the identification of catalyst deactivation, structure–activity relationships, and future opportunities is discussed.},
  author       = {Pieber, Bartholomäus},
  issn         = {1437-2096},
  journal      = {Synlett},
  number       = {1},
  pages        = {43--54},
  publisher    = {Georg Thieme Verlag},
  title        = {{Photochemical cross-couplings using semiconducting materials}},
  doi          = {10.1055/a-2690-9269},
  volume       = {37},
  year         = {2026},
}

@phdthesis{22258,
  abstract     = {Uncovering the genetic architecture of complex traits and pinpointing causal molecular drivers require the ability to distinguish true signals from noise within massive, high-dimensional omics datasets. To extract meaningful biological insights from these datasets, such as identifying causal genetic variants and proteins, scalable and accurate inference methods are essential. To this end, this thesis develops novel Bayesian inference frameworks based on Vector Approximate Message Passing and demonstrates their effectiveness in the modeling of disease onset times and quantitative physical and clinical measures.

First, we introduce gVAMP, a Bayesian framework tailored for Genome-Wide Association Studies that enables the joint modeling of quantitative complex traits across millions of genetic variants. gVAMP demonstrates superior accuracy in variable selection and out-of-sample polygenic risk prediction compared to state-of-the-art approaches. We model human height using 17 million whole-genome sequence variants from the UK Biobank, incorporating a vast number of rare variants and revealing novel associations. gVAMP achieves a prediction accuracy of approximately 46% for human height, representing the highest reported performance for this trait to date. 

Second, we present vampW, a Bayesian framework for survival analysis applied to proteomic data. By effectively handling right-censoring and complex protein dependencies within the UK Biobank Pharma Proteomics Project dataset, vampW identifies 219 protein associations across 24 disease outcomes, the majority of which are not among the top marginal discoveries. We further adjust protein levels for exponential age effects, yielding 1,308 associations and highlighting the sensitivity of the analysis to the chosen age-correction methodology. Finally, vampW improves upon the variable selection capabilities of the commonly used (penalized) variants of the Cox proportional hazards model and delivers state-of-the-art out-of-sample prediction of disease onset times.

Collectively, these methods provide powerful tools for dissecting the genetic architecture of complex traits and the proteomic drivers of disease onset. Furthermore, by delivering accurate polygenic risk scores and precise predictions of onset times, this work advances the capabilities of personalized medicine and clinical risk stratification.},
  author       = {Depope, Al},
  issn         = {2663-337X},
  keywords     = {Approximate Message Passing, GWAS, Genomics, Proteomics, Survival modeling},
  pages        = {169},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{From sparse selection to risk prediction: Approximate message passing for proteomic survival models and large-scale genomics}},
  doi          = {10.15479/AT-ISTA-22258},
  year         = {2026},
}

@article{21488,
  abstract     = {Human height is a model for the genetic analysis of complex traits, and recent studies suggest the presence of thousands of common genetic variant associations and hundreds of low-frequency/rare variants. Here, we develop a new algorithmic paradigm based on approximate message passing (genomic vector approximate message passing [gVAMP]) for identifying DNA sequence variants associated with complex traits and common diseases in large-scale whole-genome sequencing (WGS) data. We show that gVAMP accurately localizes associations to variants with the correct frequency and position in the DNA, outperforming existing fine-mapping methods in selecting the appropriate genetic variants within WGS data. We then apply gVAMP to jointly model the relationship of tens of millions of WGS variants with human height in hundreds of thousands of UK Biobank individuals. We identify 59 rare variants and gene burden scores alongside many hundreds of DNA regions containing common variant associations and show that understanding the genetic basis of complex traits will require the joint analysis of hundreds of millions of variables measured on millions of people. The polygenic risk scores obtained from gVAMP have high accuracy (including a prediction accuracy of ∼46% for human height) and outperform current methods for downstream tasks such as mixed linear model association testing across 13 UK Biobank traits. In conclusion, gVAMP offers a scalable foundation for a wider range of analyses in WGS data.},
  author       = {Depope, Al and Bajzik, Jakub and Mondelli, Marco and Robinson, Matthew Richard},
  issn         = {2666-979X},
  journal      = {Cell Genomics},
  number       = {5},
  publisher    = {Elsevier},
  title        = {{Joint modeling of whole-genome sequencing data for human height via approximate message passing}},
  doi          = {10.1016/j.xgen.2026.101162},
  volume       = {6},
  year         = {2026},
}

