@article{14949,
  abstract     = {Many approaches have been proposed to use diffusion models to augment training datasets for downstream tasks, such as classification. However, diffusion models are themselves trained on large datasets, often with noisy annotations, and it remains an open question to which extent these models contribute to downstream classification performance. In particular, it remains unclear if they generalize enough to improve over directly using the additional data of their pre-training process for augmentation. We systematically evaluate a range of existing methods to generate images from diffusion models and study new extensions to assess their benefit for data augmentation. Personalizing diffusion models towards the target data outperforms simpler prompting strategies. However, using the pre-training data of the diffusion model alone, via a simple nearest-neighbor retrieval procedure, leads to even stronger downstream performance. Our study explores the potential of diffusion models in generating new training data, and surprisingly finds that these sophisticated models are not yet able to beat a simple and strong image retrieval baseline on simple downstream vision tasks.},
  author       = {Burg, Max and Wenzel, Florian and Zietlow, Dominik and Horn, Max and Makansi, Osama and Locatello, Francesco and Russell, Chris},
  issn         = {2835-8856},
  journal      = {Journal of Machine Learning Research},
  publisher    = {ML Research Press},
  title        = {{Image retrieval outperforms diffusion models on data augmentation}},
  year         = {2023},
}

@inproceedings{14958,
  abstract     = {Causal representation learning (CRL) aims at identifying high-level causal variables from low-level data, e.g. images. Current methods usually assume that all causal variables are captured in the high-dimensional observations. In this work, we focus on learning causal representations from data under partial observability, i.e., when some of the causal variables are not observed in the measurements, and the set of masked variables changes across the different samples. We introduce some initial theoretical results for identifying causal variables under partial observability by exploiting a sparsity regularizer, focusing in particular on the linear and piecewise linear mixing function case. We provide a theorem that allows us to identify the causal variables up to permutation and element-wise linear transformations in the linear case and a lemma that allows us to identify causal variables up to linear transformation in the piecewise case. Finally, we provide a conjecture that would allow us to identify the causal variables up to permutation and element-wise linear transformations also in the piecewise linear case. We test the theorem and conjecture on simulated data, showing the effectiveness of our method.},
  author       = {Xu, Danru and Yao, Dingling and Lachapelle, Sebastien and Taslakian, Perouz and von Kügelgen, Julius and Locatello, Francesco and Magliacane, Sara},
  booktitle    = {Causal Representation Learning Workshop at NeurIPS 2023},
  location     = {New Orleans, LA, United States},
  publisher    = {OpenReview},
  title        = {{A sparsity principle for partially observable causal representation learning}},
  year         = {2023},
}

@unpublished{14961,
  abstract     = {The use of simulated data in the field of causal discovery is ubiquitous due to the scarcity of annotated real data. Recently, Reisach et al., 2021 highlighted the emergence of patterns in simulated linear data, which displays increasing marginal variance in the casual direction. As an ablation in their experiments, Montagna et al., 2023 found that similar patterns may emerge in
nonlinear models for the variance of the score vector $\nabla \log p_{\mathbf{X}}$, and introduced the ScoreSort algorithm. In this work, we formally define and characterize this score-sortability pattern of nonlinear additive noise models. We find that it defines a class of identifiable (bivariate) causal models overlapping with nonlinear additive noise models. We
theoretically demonstrate the advantages of ScoreSort in terms of statistical efficiency compared to prior state-of-the-art score matching-based methods and empirically show the score-sortability of the most common synthetic benchmarks in the literature. Our findings remark (1) the lack of diversity in the data as an important limitation in the evaluation of nonlinear causal discovery approaches, (2) the importance of thoroughly testing different settings within a problem class, and (3) the importance of analyzing statistical properties in
causal discovery, where research is often limited to defining identifiability conditions of the model. },
  author       = {Montagna, Francesco and Noceti, Nicoletta and Rosasco, Lorenzo and Locatello, Francesco},
  booktitle    = {arXiv},
  title        = {{Shortcuts for causal discovery of nonlinear models by score matching}},
  doi          = {10.48550/arXiv.2310.14246},
  year         = {2023},
}

@unpublished{14962,
  abstract     = {In this paper, we show that recent advances in video representation learning
and pre-trained vision-language models allow for substantial improvements in
self-supervised video object localization. We propose a method that first
localizes objects in videos via a slot attention approach and then assigns text
to the obtained slots. The latter is achieved by an unsupervised way to read
localized semantic information from the pre-trained CLIP model. The resulting
video object localization is entirely unsupervised apart from the implicit
annotation contained in CLIP, and it is effectively the first unsupervised
approach that yields good results on regular video benchmarks.},
  author       = {Fan, Ke and Bai, Zechen and Xiao, Tianjun and Zietlow, Dominik and Horn, Max and Zhao, Zixu and Carl-Johann Simon-Gabriel, Carl-Johann Simon-Gabriel and Shou, Mike Zheng and Locatello, Francesco and Schiele, Bernt and Brox, Thomas and Zhang, Zheng and Fu, Yanwei and He, Tong},
  booktitle    = {arXiv},
  title        = {{Unsupervised open-vocabulary object localization in videos}},
  doi          = {10.48550/arXiv.2309.09858},
  year         = {2023},
}

@unpublished{14963,
  abstract     = {Unsupervised object-centric learning methods allow the partitioning of scenes
into entities without additional localization information and are excellent
candidates for reducing the annotation burden of multiple-object tracking (MOT)
pipelines. Unfortunately, they lack two key properties: objects are often split
into parts and are not consistently tracked over time. In fact,
state-of-the-art models achieve pixel-level accuracy and temporal consistency
by relying on supervised object detection with additional ID labels for the
association through time. This paper proposes a video object-centric model for
MOT. It consists of an index-merge module that adapts the object-centric slots
into detection outputs and an object memory module that builds complete object
prototypes to handle occlusions. Benefited from object-centric learning, we
only require sparse detection labels (0%-6.25%) for object localization and
feature binding. Relying on our self-supervised
Expectation-Maximization-inspired loss for object association, our approach
requires no ID labels. Our experiments significantly narrow the gap between the
existing object-centric model and the fully supervised state-of-the-art and
outperform several unsupervised trackers.},
  author       = {Zhao, Zixu and Wang, Jiaze and Horn, Max and Ding, Yizhuo and He, Tong and Bai, Zechen and Zietlow, Dominik and Carl-Johann Simon-Gabriel, Carl-Johann Simon-Gabriel and Shuai, Bing and Tu, Zhuowen and Brox, Thomas and Schiele, Bernt and Fu, Yanwei and Locatello, Francesco and Zhang, Zheng and Xiao, Tianjun},
  booktitle    = {arXiv},
  title        = {{Object-centric multiple object tracking}},
  doi          = {10.48550/arXiv.2309.00233},
  year         = {2023},
}

@misc{14965,
  abstract     = {A method of determining a correspondence between a first biological property of a cell and one or more further biological properties of cells is provided. The first biological property and the further biological properties are determined by different analysis techniques and each are contained in a respective one of a plurality of sets of biological properties. The method includes the steps of: converting the plurality of sets of biological properties into corresponding representations in a representation format which is invariant to the technologies used to derive the biological properties; determining, in said representation format, a representation from each of the converted sets of further biological properties which most closely matches the first representation of the first biological property; and re-converting the determined representations from the representation format back to the biological properties associated with the determined representations and thereby determining a correspondence between the first biological property and each of the further biological properties.},
  author       = {Ficek, Joanna and Lehmann, Kjong-Van and Locatello, Francesco and Raetsch, Gunnar  and Stark, Stefan},
  pages        = {9},
  title        = {{Methods of determining correspondences between biological properties of cells}},
  year         = {2023},
}

@inproceedings{14974,
  abstract     = {The field of machine learning and AI has witnessed remarkable breakthroughs with the emergence of LLMs, which have also sparked a lively debate in the causal community. As researchers in this field, we are interested in exploring how LLMs relate to causality research, and how we can leverage the technology to advance it. In the second conference of Causal Learning and Reasoning (CLeaR), 2023, we held a round table discussion to gather and integrate the diverse perspectives of the CLeaR community on this topic.
There is a general consensus that LLMs are not yet capable of causal reasoning at the current
stage but has a lot of potential with public available information by CLeaR 2023. Enhancing causal machine learning is vital not only for its own sake but also to help LLMs improve their performance, especially regarding trustworthiness. In this document, we present both the summary and the raw outcome of the round table discussion. We acknowledge that with the progress of both fields, the opportunities and impact may rapidly change. We will repeat the same exercise in CLeaR 2024 to document the evolution.},
  author       = {Zhang, Cheng and Janzing, Dominik and van der Schaar, Mihaela  and Locatello, Francesco and Spirtes, Peter and Zhang, Kun and Schölkopf, Bernhard and Uhler, Caroline},
  booktitle    = {2nd Conference on Causal Learning and Reasoning},
  location     = {Tübingen, Germany},
  title        = {{Causality in the time of LLMs: Round table discussion results of CLeaR 2023}},
  year         = {2023},
}

@article{14985,
  abstract     = {Lead sulfide (PbS) presents large potential in thermoelectric application due to its earth-abundant S element. However, its inferior average ZT (ZTave) value makes PbS less competitive with its analogs PbTe and PbSe. To promote its thermoelectric performance, this study implements strategies of continuous Se alloying and Cu interstitial doping to synergistically tune thermal and electrical transport properties in n-type PbS. First, the lattice parameter of 5.93 Å in PbS is linearly expanded to 6.03 Å in PbS0.5Se0.5 with increasing Se alloying content. This expanded lattice in Se-alloyed PbS not only intensifies phonon scattering but also facilitates the formation of Cu interstitials. Based on the PbS0.6Se0.4 content with the minimal lattice thermal conductivity, Cu interstitials are introduced to improve the electron density, thus boosting the peak power factor, from 3.88 μW cm−1 K−2 in PbS0.6Se0.4 to 20.58 μW cm−1 K−2 in PbS0.6Se0.4−1%Cu. Meanwhile, the lattice thermal conductivity in PbS0.6Se0.4−x%Cu (x = 0–2) is further suppressed due to the strong strain field caused by Cu interstitials. Finally, with the lowered thermal conductivity and high electrical transport properties, a peak ZT ~1.1 and ZTave ~0.82 can be achieved in PbS0.6Se0.4 − 1%Cu at 300–773K, which outperforms previously reported n-type PbS.},
  author       = {Liu, Zhengtao and Hong, Tao and Xu, Liqing and Wang, Sining and Gao, Xiang and Chang, Cheng and Ding, Xiangdong and Xiao, Yu and Zhao, Li‐Dong},
  issn         = {2767-441X},
  journal      = {Interdisciplinary Materials},
  number       = {1},
  pages        = {161--170},
  publisher    = {Wiley},
  title        = {{Lattice expansion enables interstitial doping to achieve a high average ZT in n‐type PbS}},
  doi          = {10.1002/idm2.12056},
  volume       = {2},
  year         = {2023},
}

@inproceedings{14989,
  abstract     = {Encryption alone is not enough for secure end-to end encrypted messaging: a server must also honestly serve public keys to users. Key transparency has been presented as an efficient
solution for detecting (and hence deterring) a server that attempts to dishonestly serve keys. Key transparency involves two major components: (1) a username to public key mapping, stored and cryptographically committed to by the server, and, (2) an outof-band consistency protocol for serving short commitments to users. In the setting of real-world deployments and supporting production scale, new challenges must be considered for both of these components. We enumerate these challenges and provide solutions to address them. In particular, we design and implement a memory-optimized and privacy-preserving verifiable data structure for committing to the username to public key store.
To make this implementation viable for production, we also integrate support for persistent and distributed storage. We also propose a future-facing solution, termed “compaction”, as
a mechanism for mitigating practical issues that arise from dealing with infinitely growing server data structures. Finally, we implement a consensusless solution that achieves the minimum requirements for a service that consistently distributes commitments for a transparency application, providing a much more efficient protocol for distributing small and consistent
commitments to users. This culminates in our production-grade implementation of a key transparency system (Parakeet) which we have open-sourced, along with a demonstration of feasibility through our benchmarks.},
  author       = {Malvai, Harjasleen and Kokoris Kogias, Eleftherios and Sonnino, Alberto and Ghosh, Esha and Oztürk, Ercan and Lewi, Kevin and Lawlor, Sean},
  booktitle    = {Proceedings of the 2023 Network and Distributed System Security Symposium},
  isbn         = {1891562835},
  location     = {San Diego, CA, United States},
  publisher    = {Internet Society},
  title        = {{Parakeet: Practical key transparency for end-to-end eEncrypted messaging}},
  doi          = {10.14722/ndss.2023.24545},
  year         = {2023},
}

@misc{14990,
  abstract     = {The software artefact to evaluate the approximation of stationary distributions implementation.},
  author       = {Meggendorfer, Tobias},
  publisher    = {Zenodo},
  title        = {{Artefact for: Correct Approximation of Stationary Distributions}},
  doi          = {10.5281/ZENODO.7548214},
  year         = {2023},
}

@misc{14991,
  abstract     = {This repository contains the data, scripts, WRF codes and files required to reproduce the results of the manuscript "Assessing Memory in Convection Schemes Using Idealized Tests" submitted to the Journal of Advances in Modeling Earth Systems (JAMES).},
  author       = {Hwong, Yi-Ling and Colin, Maxime and Aglas, Philipp and Muller, Caroline J and Sherwood, Steven C.},
  publisher    = {Zenodo},
  title        = {{Data-assessing memory in convection schemes using idealized tests}},
  doi          = {10.5281/ZENODO.7757041},
  year         = {2023},
}

@inbook{14992,
  abstract     = {In this chapter we first review the Levy–Lieb functional, which gives the lowest kinetic and interaction energy that can be reached with all possible quantum states having a given density. We discuss two possible convex generalizations of this functional, corresponding to using mixed canonical and grand-canonical states, respectively. We present some recent works about the local density approximation, in which the functionals get replaced by purely local functionals constructed using the uniform electron gas energy per unit volume. We then review the known upper and lower bounds on the Levy–Lieb functionals. We start with the kinetic energy alone, then turn to the classical interaction alone, before we are able to put everything together. A later section is devoted to the Hohenberg–Kohn theorem and the role of many-body unique continuation in its proof.},
  author       = {Lewin, Mathieu and Lieb, Elliott H. and Seiringer, Robert},
  booktitle    = {Density Functional Theory},
  editor       = {Cances, Eric and Friesecke, Gero},
  isbn         = {9783031223396},
  issn         = {3005-0286},
  pages        = {115--182},
  publisher    = {Springer},
  title        = {{Universal Functionals in Density Functional Theory}},
  doi          = {10.1007/978-3-031-22340-2_3},
  year         = {2023},
}

@inproceedings{14993,
  abstract     = {Traditional top-down approaches for global health have historically failed to achieve social progress (Hoffman et al., 2015; Hoffman & Røttingen, 2015). Recently, however, a more holistic, multi-level approach termed One Health (OH) (Osterhaus et al., 2020) is being adopted. Several sets of challenges have been identified for the implementation of OH (dos S. Ribeiro et al., 2019), including policy and funding, education and training, and multi-actor, multi-domain, and multi-level collaborations. These exist despite the increasing accessibility to
knowledge and digital collaborative research tools through the internet. To address some of these challenges, we propose a general framework for grassroots community-based means of participatory research. Additionally, we present a specific roadmap to create a Machine Learning for Global Health community in Africa. The proposed framework aims to enable any small group of individuals with scarce resources to build and sustain an online community within approximately two years. We provide a discussion on the potential impact of the proposed framework for global health research collaborations.},
  author       = {Currin, Christopher and Asiedu , Mercy Nyamewaa and Fourie, Chris and Rosman, Benjamin and Turki, Houcemeddine and Lambebo Tonja, Atnafu and Abbott, Jade and Ajala, Marvellous and Adedayo, Sadiq Adewale and Emezue, Chris Chinenye and Machangara, Daphne},
  booktitle    = {1st Workshop on Machine Learning & Global Health},
  location     = {Kigali, Rwanda},
  publisher    = {OpenReview},
  title        = {{A framework for grassroots research collaboration in machine learning and global health}},
  year         = {2023},
}

@misc{14994,
  abstract     = {This resource contains the artifacts for reproducing the experimental results presented in the paper titled "A Flexible Toolchain for Symbolic Rabin Games under Fair and Stochastic Uncertainties" that has been submitted in CAV 2023.},
  author       = {Majumdar, Rupak and Mallik, Kaushik and Rychlicki, Mateusz and Schmuck, Anne-Kathrin and Soudjani, Sadegh},
  publisher    = {Zenodo},
  title        = {{A flexible toolchain for symbolic rabin games under fair and stochastic uncertainties}},
  doi          = {10.5281/ZENODO.7877790},
  year         = {2023},
}

@misc{15027,
  abstract     = {This data repository underpins the paper, published in PNAS (doi pending) and bioarxiv (doi: https://doi.org/10.1101/2023.07.05.547777).},
  author       = {Curk, Samo},
  publisher    = {Figshare},
  title        = {{aggregation_data}},
  year         = {2023},
}

@misc{15035,
  abstract     = {This artifact aims to reproduce experiments from the paper Monitoring Hyperproperties With Prefix Transducers accepted at RV'23, and give further pointers to implementation of prefix transducers.
It has two parts: a pre-compiled docker image and sources that one can use to compile (locally or in docker) the software and run the experiments.},
  author       = {Chalupa, Marek and Henzinger, Thomas A},
  publisher    = {Zenodo},
  title        = {{Monitoring hyperproperties with prefix transducers}},
  doi          = {10.5281/ZENODO.8191723},
  year         = {2023},
}

@article{15173,
  abstract     = {We show that the number of linear spaces on a set of n points and the number of rank-3 matroids on a ground set of size n are both of the form (cn+o(n))n2/6, where c=e3√/2−3(1+3–√)/2. This is the final piece of the puzzle for enumerating fixed-rank matroids at this level of accuracy: the numbers of rank-1 and rank-2 matroids on a ground set of size n have exact representations in terms of well-known combinatorial functions, and it was recently proved by van der Hofstad, Pendavingh, and van der Pol that for constant r≥4 there are (e1−rn+o(n))nr−1/r! rank-r matroids on a ground set of size n. In our proof, we introduce a new approach for bounding the number of clique decompositions of a complete graph, using quasirandomness instead of the so-called entropy method that is common in this area.},
  author       = {Kwan, Matthew Alan and Sah, Ashwin and Sawhney, Mehtaab},
  issn         = {1778-3569},
  journal      = {Comptes Rendus Mathematique},
  number       = {G2},
  pages        = {565--575},
  publisher    = {Academie des Sciences},
  title        = {{Enumerating matroids and linear spaces}},
  doi          = {10.5802/crmath.423},
  volume       = {361},
  year         = {2023},
}

@misc{15292,
  abstract     = {We present a rigid body animation technique which prevents solids from interpenetrating, dissipates energy through friction, and propagates shocks through contacts. We employ the Alternating Direction Method of Multipliers (ADMM) to couple non-smooth Coulomb friction with impact propagation, allowing efficient and accurate non-smooth dynamics along with a correct transmission of impacts through assemblies of rigid bodies. We further extend our method to model adhesion, dynamic friction and lubricated contact.},
  author       = {Chen, Yi-Lu and Ly, Mickaël and Wojtan, Christopher J},
  booktitle    = {Proceedings of the ACM SIGGRAPH/Eurographics Symposium on Computer Animation},
  location     = {Los Angeles, CA, United States},
  publisher    = {ACM},
  title        = {{Unified treatment of contact, friction and shock-propagation in rigid body animation}},
  doi          = {10.1145/3606037.3606836},
  year         = {2023},
}

@article{17074,
  abstract     = {We verify Bogoliubov's approximation for translation invariant Bose gases in the mean field regime, i.e. we prove that the ground state energy EN is given by EN=NeH+infσ(H)+oN→∞(1), where N is the number of particles, eH is the minimal Hartree energy and H is the Bogoliubov Hamiltonian. As an intermediate result we show the existence of approximate ground states ΨN, i.e. states satisfying ⟨HN⟩ΨN=EN+oN→∞(1), exhibiting complete Bose--Einstein condensation with respect to one of the Hartree minimizers.},
  author       = {Brooks, Morris and Seiringer, Robert},
  issn         = {2690-1005},
  journal      = {Probability and Mathematical Physics},
  number       = {4},
  pages        = {939--1000},
  publisher    = {Mathematical Sciences Publishers},
  title        = {{Validity of Bogoliubov’s approximation fortranslation-invariant Bose gases}},
  doi          = {10.2140/pmp.2022.3.939},
  volume       = {3},
  year         = {2023},
}

@article{17078,
  abstract     = {For the emergence of life, the abiotic synthesis of RNA from its monomers is a central step. We found that in alkaline, drying conditions in bulk and at heated air‐water interfaces, 2′,3′‐cyclic nucleotides oligomerised without additional catalyst, forming up to 10‐mers within a day. The oligomerisation proceeded at a pH range of 7–12, at temperatures between 40–80 °C and was marginally enhanced by K<jats:sup>+</jats:sup> ions. Among the canonical ribonucleotides, cGMP oligomerised most efficiently. Quantification was performed using HPLC coupled to ESI‐TOF by fitting the isotope distribution to the mass spectra. Our study suggests a oligomerisation mechanism where cGMP aids the incorporation of the relatively unreactive nucleotides C, A and U. The 2′,3′‐cyclic ribonucleotides are byproducts of prebiotic phosphorylation, nucleotide syntheses and RNA hydrolysis, indicating direct recycling pathways. The simple reaction condition offers a plausible entry point for RNA to the evolution of life on early Earth.},
  author       = {Dass, Avinash Vicholous and Wunnava, Sreekar and Langlais, Juliette and von der Esch, Beatriz and Krusche, Maik and Ufer, Lennard and Chrisam, Nico and Dubini, Romeo C. A. and Gartner, Florian and Angerpointner, Severin and Dirscherl, Christina F. and Rovo, Petra and Mast, Christof B. and Šponer, Judit E. and Ochsenfeld, Christian and Frey, Erwin and Braun, Dieter},
  issn         = {2570-4206},
  journal      = {ChemSystemsChem},
  number       = {1},
  publisher    = {Wiley},
  title        = {{RNA oligomerisation without added catalyst from 2′,3′‐cyclic nucleotides by drying at air-water interfaces}},
  doi          = {10.1002/syst.202200026},
  volume       = {5},
  year         = {2023},
}

