@article{18653,
  abstract     = {Charge sensing is a sensitive technique for probing quantum devices, of particular importance for spin-qubit readout. To achieve good readout sensitivities, the proximity of the charge sensor to the device to be measured is a necessity. However, this proximity also means that the operation of the device affects, in turn, the sensor tuning and ultimately the readout sensitivity. We present an approach for compensating for this crosstalk effect allowing for the gate voltages of the measured device to be swept in a 1-V × 1-V window while maintaining a sensor configuration chosen by a Bayesian optimizer. Our algorithm will hopefully be a major contribution to the suite of fully automated solutions required for the operation of large quantum device architectures.},
  author       = {Hickie, Joseph and Van Straaten, Barnaby and Fedele, Federico and Jirovec, Daniel and Ballabio, Andrea and Chrastina, Daniel and Isella, Giovanni and Katsaros, Georgios and Ares, Natalia},
  issn         = {2331-7019},
  journal      = {Physical Review Applied},
  number       = {6},
  publisher    = {American Physical Society},
  title        = {{Automated long-range compensation of an rf quantum dot sensor}},
  doi          = {10.1103/PhysRevApplied.22.064026},
  volume       = {22},
  year         = {2024},
}

@article{18654,
  abstract     = {We compute the rotational anisotropy of the free energy of 𝛼−RuCl3 in an external magnetic field. This quantity, known as the magnetotropic susceptibility, 𝑘, relates to the second derivative of the free energy with respect to the angle of rotation. We have used approximation-free, auxiliary-field quantum Monte Carlo simulations for a realistic model of 𝛼−RuCl3 and optimized the path integral to alleviate the negative sign problem. This allows us to reach temperatures down to 30K—an energy scale below the dominant Kitaev coupling. We demonstrate that the magnetotropic spin susceptibility in this model of 𝛼−RuCl3 displays scaling behavior 𝑘=𝑇⁢𝑓⁡(𝐵/𝑇) at high temperatures. Once the uniform susceptibility departs from the Curie law (i.e., at the energy scale of the exchange interactions), it appears to transition to an emergent scalinglike behavior, characterized by a different function 𝑓 at lower temperatures, stemming from the locality of torque fluctuations. We observe a remarkable numerical match between experiment and simulations and we also find qualitative agreement with the pure Kitaev model. In comparison, for the XXZ Heisenberg Hamiltonian, the scaling 𝑘=𝑇⁢𝑓⁡(𝐵/𝑇) breaks down at a temperature scale where the uniform spin susceptibility deviates from the Curie law and never reemerges at low temperatures.},
  author       = {Sato, Toshihiro and Ramshaw, B. J. and Modic, Kimberly A and Assaad, Fakher F.},
  issn         = {2469-9969},
  journal      = {Physical Review B},
  number       = {20},
  publisher    = {American Physical Society},
  title        = {{Scale-invariant magnetic anisotropy in α-RuCl3: A quantum Monte Carlo study}},
  doi          = {10.1103/PhysRevB.110.L201114},
  volume       = {110},
  year         = {2024},
}

@article{18655,
  abstract     = {Let Qd be the d-dimensional binary hypercube. We say that P={v1,…,vk} is an increasing path of length k−1 in Qd, if for every i∈[k−1] the edge vivi+1 is obtained by switching some zero coordinate in vi to a one coordinate in vi+1.
Form a random subgraph Qdp by retaining each edge in E(Qd) independently with probability p. We show that there is a phase transition with respect to the length of a longest increasing path around p=ed. Let α be a constant and let p=αd. When α<e, then there exists a δ∈[0,1) such that whp a longest increasing path in Qdp is of length at most δd. On the other hand, when α>e, whp there is a path of length d−2 in Qdp, and in fact, whether it is of length d−2,d−1, or d depends on whether the all-zero and all-one vertices percolate or not.},
  author       = {Anastos, Michael and Diskin, Sahar and Elboim, Dor and Krivelevich, Michael},
  issn         = {1083-589X},
  journal      = {Electronic Communications in Probability},
  publisher    = {Duke University Press},
  title        = {{Climbing up a random subgraph of the hypercube}},
  doi          = {10.1214/24-ECP639},
  volume       = {29},
  year         = {2024},
}

@unpublished{18673,
  abstract     = {Motivated by applications to crystalline materials, we generalize the merge tree and the related barcode of a filtered complex to the periodic setting in Euclidean space. They are invariant under isometries, changing bases, and indeed changing lattices. In addition, we prove stability under perturbations and provide an algorithm that under mild geometric conditions typically satisfied by crystalline materials takes O((n+m)logn) time, in which n and m are the numbers of vertices and edges in the quotient complex, respectively.
},
  author       = {Edelsbrunner, Herbert and Heiss, Teresa},
  booktitle    = {arXiv},
  title        = {{Merge trees of periodic filtrations}},
  doi          = {10.48550/arXiv.2408.16575},
  year         = {2024},
}

@phdthesis{18674,
  abstract     = {Mapping the complex and dense arrangement of cells and their connectivity in brain tissue requires volumetric imaging at nanoscale spatial resolution. While light microscopy excels at visualizing specific molecules and individual cells, achieving dense, synapse-level circuit reconstruction has not been possible with any light microscopy technique. Thus, the goal of my work was to develop image and data analysis pipelines for brain tissue visualization and reconstruction with light microscopy. To achieve dense circuit reconstruction with single-synapse resolution, I developed both conventional and deep-learning-based synapse detection algorithms, as well as connectivity analysis pipelines that integrate synapse detection with volumetric segmentation of brain tissue.},
  author       = {Lyudchik, Julia},
  isbn         = { 978-3-99078-051-0},
  issn         = {2663-337X},
  pages        = {217},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Image analysis for brain tissue reconstruction with super-resolution light microscopy}},
  doi          = {10.15479/at:ista:18674},
  year         = {2024},
}

@article{18709,
  abstract     = {We measure the mass distribution of main-sequence (MS) companions to hot subdwarf B stars (sdBs) in post-common envelope binaries (PCEBs). We carried out a spectroscopic survey of 14 eclipsing systems ("HW Vir binaries") with orbital periods of 3.8 < Porb < 12 hr, resulting in a well-understood selection function and a near-complete sample of HW Vir binaries with G < 16. We constrain companion masses from the radial velocity curves of the sdB stars. The companion mass distribution peaks at MMS ≈ 0.15 M⊙ and drops off at MMS > 0.2 M⊙, with only two systems hosting companions above the fully convective limit. There is no correlation between Porb and MMS within the sample. A similar drop-off in the companion mass distribution of white dwarf (WD) + MS PCEBs has been attributed to disrupted magnetic braking (MB) below the fully convective limit. We compare the sdB companion mass distribution to predictions of binary evolution simulations with a range of MB laws. Because sdBs have short lifetimes compared to WDs, explaining the lack of higher-mass MS companions to sdBs with disrupted MB requires MB to be boosted by a factor of 20–100 relative to MB laws inferred from the rotation evolution of single stars. We speculate that such boosting may be a result of irradiation-driven enhancement of the MS stars' winds. An alternative possibility is that common envelope evolution favors low-mass companions in short-period orbits, but the existence of massive WD companions to sdBs with similar periods disfavors this scenario.},
  author       = {Blomberg, Lisa and El-Badry, Kareem and Breivik, Katelyn and Caiazzo, Ilaria and Nagarajan, Pranav and Rodriguez, Antonio and Van Roestel, Jan and Vanderbosch, Zachary P. and Yamaguchi, Natsuko},
  issn         = {0004-6280},
  journal      = {Publications of the Astronomical Society of the Pacific},
  number       = {12},
  publisher    = {IOP Publishing},
  title        = {{The companion mass distribution of post common envelope hot subdwarf binaries: Evidence for boosted and disrupted magnetic braking?}},
  doi          = {10.1088/1538-3873/ad94a2},
  volume       = {136},
  year         = {2024},
}

@misc{18716,
  abstract     = {Data for publication 10.1039/d4cp03727h},
  author       = {Hrast, Mateja},
  publisher    = {Zenodo},
  title        = {{Data for: Ab initio Auger spectrum of the ultrafast dissociating 2p3/2−1σ* resonance in HCl}},
  doi          = {10.5281/ZENODO.13833474},
  year         = {2024},
}

@inproceedings{18755,
  abstract     = {A universalthresholdizer (UT), constructed from a threshold fully homomorphic encryption by Boneh et. al , Crypto 2018, is a general framework for universally thresholdizing many cryptographic schemes. However, their framework is insufficient to construct strongly secure threshold schemes, such as threshold signatures and threshold public-key encryption, etc.

In this paper, we strengthen the security definition for a universal thresholdizer and propose a scheme which satisfies our stronger security notion. Our UT scheme is an improvement of Boneh et. al ’s construction at the level of threshold fully homomorphic encryption using a key homomorphic pseudorandom function. We apply our strongly secure UT scheme to construct strongly secure threshold signatures and threshold public-key encryption.},
  author       = {Ebrahimi, Ehsan and Yadav, Anshu},
  booktitle    = {30th International Conference on the Theory and Application of Cryptology and Information Security},
  isbn         = {9789819608904},
  issn         = {1611-3349},
  location     = {Kolkata, India},
  pages        = {207--239},
  publisher    = {Springer Nature},
  title        = {{Strongly secure universal thresholdizer}},
  doi          = {10.1007/978-981-96-0891-1_7},
  volume       = {15486},
  year         = {2024},
}

@inproceedings{18756,
  abstract     = {The evasive LWE assumption, proposed by Wee [Eurocrypt’22 Wee] for constructing a lattice-based optimal broadcast encryption, has shown to be a powerful assumption, adopted by subsequent works to construct advanced primitives ranging from ABE variants to obfuscation for null circuits. However, a closer look reveals significant differences among the precise assumption statements involved in different works, leading to the fundamental question of how these assumptions compare to each other. In this work, we initiate a more systematic study on evasive LWE assumptions:
(i) Based on the standard LWE assumption, we construct simple counterexamples against three private-coin evasive LWE variants, used in [Crypto’22 Tsabary, Asiacrypt’22 VWW, Crypto’23 ARYY] respectively, showing that these assumptions are unlikely to hold.

(ii) Based on existing evasive LWE variants and our counterexamples, we propose and define three classes of plausible evasive LWE assumptions, suitably capturing all existing variants for which we are not aware of non-obfuscation-based counterexamples.

(iii) We show that under our assumption formulations, the security proofs of [Asiacrypt’22 VWW] and [Crypto’23 ARYY] can be recovered, and we reason why the security proof of [Crypto’22 Tsabary] is also plausibly repairable using an appropriate evasive LWE assumption.},
  author       = {Brzuska, Chris and Ünal, Akin and Woo, Ivy K.Y.},
  booktitle    = {30th International Conference on the Theory and Application of Cryptology and Information Security},
  isbn         = {9789819608935},
  issn         = {1611-3349},
  location     = {Kolkata, India},
  pages        = {418--449},
  publisher    = {Springer Nature},
  title        = {{Evasive LWE assumptions: Definitions, classes, and counterexamples}},
  doi          = {10.1007/978-981-96-0894-2_14},
  volume       = {15487},
  year         = {2024},
}

@inproceedings{18758,
  abstract     = {MaxCut is a classical NP-complete problem and a crucial building block in many combinatorial algorithms. The famous Edwards-Erdős bound states that any connected graph on n vertices with m edges contains a cut of size at least m/2+(n-1)/4. Crowston, Jones and Mnich [Algorithmica, 2015] showed that the MaxCut problem on simple connected graphs admits an FPT algorithm, where the parameter k is the difference between the desired cut size c and the lower bound given by the Edwards-Erdős bound. This was later improved by Etscheid and Mnich [Algorithmica, 2017] to run in parameterized linear time, i.e., f(k)⋅ O(m). We improve upon this result in two ways: Firstly, we extend the algorithm to work also for multigraphs (alternatively, graphs with positive integer weights). Secondly, we change the parameter; instead of the difference to the Edwards-Erdős bound, we use the difference to the Poljak-Turzík bound. The Poljak-Turzík bound states that any weighted graph G has a cut of size at least (w(G))/2+(w_MSF(G))/4, where w(G) denotes the total weight of G, and w_MSF(G) denotes the weight of its minimum spanning forest. In connected simple graphs the two bounds are equivalent, but for multigraphs the Poljak-Turzík bound can be larger and thus yield a smaller parameter k. Our algorithm also runs in parameterized linear time, i.e., f(k)⋅ O(m+n).},
  author       = {Lill, Jonas and Petrova, Kalina H and Weber, Simon},
  booktitle    = {19th International Symposium on Parameterized and Exact Computation},
  isbn         = {9783959773539},
  issn         = {1868-8969},
  location     = {Egham, United Kingdom},
  publisher    = {Schloss Dagstuhl - Leibniz-Zentrum für Informatik},
  title        = {{Linear-time MaxCut in multigraphs parameterized above the Poljak-Turzík bound}},
  doi          = {10.4230/LIPIcs.IPEC.2024.2},
  volume       = {321},
  year         = {2024},
}

@article{18760,
  abstract     = {With the remarkable sensitivity and resolution of JWST in the infrared, measuring rest-optical kinematics of galaxies at z > 5 has become possible for the first time. This study pilots a new method for measuring galaxy dynamics for highly multiplexed, unbiased samples by combining FRESCO NIRCam grism spectroscopy and JADES medium-band imaging. Here we present one of the first JWST kinematic measurements for a galaxy at z > 5. We find a significant velocity gradient, which, if interpreted as rotation, yields Vrot = 305 ± 70 km s−1, and we hence refer to this galaxy as Twister-z5. With a rest-frame optical effective radius of re = 2.25 kpc, the high rotation velocity in this galaxy is not due to a compact size, as may be expected in the early Universe, but rather to a high total mass, (math formula). This is a factor of roughly 10× higher than the stellar mass within re. We also observe that the radial Hα equivalent width profile and the specific star formation rate map from resolved stellar population modeling are centrally depressed by a factor of ∼1.5 from the center to re. Combined with the morphology of the line-emitting gas in comparison to the continuum, this centrally suppressed star formation is consistent with a star-forming disk surrounding a bulge growing inside out. While large, rapidly rotating disks are common to z ∼ 2, the existence of one after only 1 Gyr of cosmic time, shown for the first time in ionized gas, adds to the growing evidence that some galaxies matured earlier than expected in the history of the Universe.},
  author       = {Nelson, Erica and Brammer, Gabriel and Giménez-Arteaga, Clara and Oesch, Pascal A. and Naidu, Rohan P. and Übler, Hannah and Matharu, Jasleen and Shapley, Alice E. and Whitaker, Katherine E. and Wisnioski, Emily and Förster Schreiber, Natascha M. and Smit, Renske and Van Dokkum, Pieter and Chisholm, John and Endsley, Ryan and Hartley, Abigail I. and Gibson, Justus and Giovinazzo, Emma and Illingworth, Garth and Labbe, Ivo and Maseda, Michael V. and Matthee, Jorryt J and Covelo Paz, Alba and Price, Sedona H. and Reddy, Naveen A. and Shivaei, Irene and Weibel, Andrea and Wuyts, Stijn and Xiao, Mengyuan and Alberts, Stacey and Baker, William M. and Bunker, Andrew J. and Cameron, Alex J. and Charlot, Stephane and Eisenstein, Daniel J. and De Graaff, Anna and Ji, Zhiyuan and Johnson, Benjamin D. and Jones, Gareth C. and Maiolino, Roberto and Robertson, Brant and Sandles, Lester and Suess, Katherine A. and Tacchella, Sandro and Williams, Christina C. and Witstok, Joris},
  issn         = {2041-8213},
  journal      = {Astrophysical Journal Letters},
  number       = {2},
  publisher    = {IOP Publishing},
  title        = {{Ionized gas kinematics with FRESCO: An extended, massive, rapidly rotating galaxy at z = 5.4}},
  doi          = {10.3847/2041-8213/ad7b17},
  volume       = {976},
  year         = {2024},
}

@article{18762,
  abstract     = {Consider the random variable $\mathrm{Tr}( f_1(W)A_1\dots f_k(W)A_k)$ where $W$ is an $N\times N$ Hermitian Wigner matrix, $k\in\mathbb{N}$, and choose (possibly $N$-dependent) regular functions $f_1,\dots, f_k$ as well as bounded deterministic matrices $A_1,\dots,A_k$. We give a functional central limit theorem showing that the fluctuations around the expectation are Gaussian. Moreover, we determine the limiting covariance structure and give explicit error bounds in terms of the scaling of $f_1,\dots,f_k$ and the number of traceless matrices among $A_1,\dots,A_k$, thus extending the results of [Cipolloni, Erdős, Schröder 2023] to products of arbitrary length $k\geq2$. As an application, we consider the fluctuation of $\mathrm{Tr}(\mathrm{e}^{\mathrm{i} tW}A_1\mathrm{e}^{-\mathrm{i} tW}A_2)$ around its thermal value $\mathrm{Tr}(A_1)\mathrm{Tr}(A_2)$ when $t$ is large and give an explicit formula for the variance.},
  author       = {Reker, Jana},
  issn         = {1083-6489},
  journal      = {Electronic Journal of Probability},
  publisher    = {Institute of Mathematical Statistics},
  title        = {{Multi-point functional central limit theorem for Wigner matrices}},
  doi          = {10.1214/24-EJP1247},
  volume       = {29},
  year         = {2024},
}

@article{18779,
  abstract     = {Unsupervised segmentation in biological and non-biological images is only partially resolved. Segmentation either requires arbitrary thresholds or large teaching datasets. Here, we propose a spatial autocorrelation method based on Local Moran’s <jats:italic>I</jats:italic> coefficient to differentiate signal, background, and noise in any type of image. The method, originally described for geoinformatics, does not require a predefined intensity threshold or teaching algorithm for image segmentation and allows quantitative comparison of samples obtained in different conditions. It utilizes relative intensity as well as spatial information of neighboring elements to select spatially contiguous groups of pixels. We demonstrate that Moran’s method outperforms threshold-based method in both artificially generated as well as in natural images especially when background noise is substantial. This superior performance can be attributed to the exclusion of false positive pixels resulting from isolated, high intensity pixels in high noise conditions. To test the method’s power in real situation, we used high power confocal images of the somatosensory thalamus immunostained for Kv4.2 and Kv4.3 (A-type) voltage-gated potassium channels in mice. Moran’s method identified high-intensity Kv4.2 and Kv4.3 ion channel clusters in the thalamic neuropil. Spatial distribution of these clusters displayed strong correlation with large sensory axon terminals of subcortical origin. The unique association of the special presynaptic terminals and a postsynaptic voltage-gated ion channel cluster was confirmed with electron microscopy. These data demonstrate that Moran’s method is a rapid, simple image segmentation method optimal for variable and high noise conditions.},
  author       = {Dávid, Csaba and Giber, Kristóf and Szigeti, Margit Katalin and Köllő, Mihály and Nusser, Zoltan and Acsady, Laszlo},
  issn         = {2050-084X},
  journal      = {eLife},
  publisher    = {eLife Sciences Publications},
  title        = {{A novel image segmentation method based on spatial autocorrelation identifies A-type potassium channel clusters in the thalamus}},
  doi          = {10.7554/elife.89361},
  volume       = {12},
  year         = {2024},
}

@inproceedings{18847,
  abstract     = {Machine Learning and AI have the potential to transform data-driven
scientific discovery, enabling accurate predictions for several scientific
phenomena. As many scientific questions are inherently causal, this paper looks
at the causal inference task of treatment effect estimation, where the outcome
of interest is recorded in high-dimensional observations in a Randomized
Controlled Trial (RCT). Despite being the simplest possible causal setting and
a perfect fit for deep learning, we theoretically find that many common choices
in the literature may lead to biased estimates. To test the practical impact of
these considerations, we recorded ISTAnt, the first real-world benchmark for
causal inference downstream tasks on high-dimensional observations as an RCT
studying how garden ants (Lasius neglectus) respond to microparticles applied
onto their colony members by hygienic grooming. Comparing 6 480 models
fine-tuned from state-of-the-art visual backbones, we find that the sampling
and modeling choices significantly affect the accuracy of the causal estimate,
and that classification accuracy is not a proxy thereof. We further validated
the analysis, repeating it on a synthetically generated visual data set
controlling the causal model. Our results suggest that future benchmarks should
carefully consider real downstream scientific questions, especially causal
ones. Further, we highlight guidelines for representation learning methods to
help answer causal questions in the sciences.},
  author       = {Cadei, Riccardo and Lindorfer, Lukas and Cremer, Sylvia and Schmid, Cordelia and Locatello, Francesco},
  booktitle    = {ICML 2024 Workshop AI4Science},
  publisher    = {Curran Associates},
  title        = {{Smoke and mirrors in causal downstream tasks}},
  volume       = {38},
  year         = {2024},
}

@article{18856,
  abstract     = {This research is aimed to solve the tweet/user geolocation prediction task and provide a flexible methodology for the geo-tagging of textual big data. The suggested approach implements neural networks for natural language processing (NLP) to estimate the location as coordinate pairs (longitude, latitude) and two-dimensional Gaussian Mixture Models (GMMs). The scope of proposed models has been finetuned on a Twitter dataset using pretrained Bidirectional Encoder Representations from Transformers (BERT) as base models. Performance metrics show a median error of fewer than 30 km on a worldwide-level, and fewer than 15 km on the US-level datasets for the models trained and evaluated on text features of tweets' content and metadata context. Our source code and data are available at https://github.com/K4TEL/geo-twitter.git.},
  author       = {Lutsai, Kateryna and Lampert, Christoph},
  issn         = {1948-660X},
  journal      = {Journal of Spatial Information Science},
  number       = {29},
  pages        = {69--99},
  publisher    = {University of Maine},
  title        = {{Predicting the geolocation of tweets using transformer models on customized data}},
  doi          = {10.5311/JOSIS.2024.29.295},
  year         = {2024},
}

@article{18868,
  abstract     = {We develop two new highly efficient estimators to measure the polarization (Stokes parameters) in experiments that constrain the position angle of individual photons such as scattering and gas-pixel-detector polarimeters, and analyse in detail a previously proposed estimator. All three of these estimators are at least fifty percent more efficient on typical datasets than the standard estimator used in the field. We present analytic estimates of the variance of these estimators and numerical experiments to verify these estimates. Two of the three estimators can be calculated quickly and directly through summations over the measurements of individual photons.},
  author       = {Heyl, Jeremy and González-Caniulef, Denis and Caiazzo, Ilaria},
  issn         = {2565-6120},
  journal      = {The Open Journal of Astrophysics},
  publisher    = {Maynooth Academic Publishing},
  title        = {{Optimal summary statistics for X-ray polarization}},
  doi          = {10.33232/001c.117476},
  volume       = {7},
  year         = {2024},
}

@inproceedings{18875,
  abstract     = {Current state-of-the-art methods for differentially private model training are based on matrix factorization techniques. However, these methods suffer from high computational overhead because they require numerically solving a demanding optimization problem to determine an approximately optimal factorization prior to the actual model training. In this work, we present a new matrix factorization approach, BSR, which overcomes this computational bottleneck. By exploiting properties of the standard matrix square root, BSR allows to efficiently handle also large-scale problems. For the key scenario of stochastic gradient descent with momentum and weight decay, we even derive analytical expressions for BSR that render the computational overhead negligible. We prove bounds on the approximation quality that hold both in the centralized and in the federated learning setting. Our numerical experiments demonstrate that models trained using BSR perform on par with the best existing methods, while completely avoiding their computational overhead.},
  author       = {Kalinin, Nikita and Lampert, Christoph},
  booktitle    = {38th Annual Conference on Neural Information Processing Systems},
  issn         = {1049-5258},
  location     = {Vancouver, Canada},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{Banded square root matrix factorization for differentially private model training}},
  volume       = {37},
  year         = {2024},
}

@inproceedings{18890,
  abstract     = {Deep Neural Collapse (DNC) refers to the surprisingly rigid structure of the data representations in the final layers of Deep Neural Networks (DNNs). Though the phenomenon has been measured in a variety of settings, its emergence is typically explained via data-agnostic approaches, such as the unconstrained features model. In this work, we introduce a data-dependent setting where DNC forms due to feature learning through the average gradient outer product (AGOP). The AGOP is defined with respect to a learned predictor and is equal to the uncentered covariance matrix of its input-output gradients averaged over the training dataset. The Deep Recursive Feature Machine (Deep RFM) is a method that constructs a neural network by iteratively mapping the data with the AGOP and applying an untrained random feature map. We demonstrate empirically that DNC occurs in Deep RFM across standard settings as a consequence of the projection with the AGOP matrix computed at each layer. Further, we theoretically explain DNC in Deep RFM in an asymptotic setting and as a result of kernel learning. We then provide evidence that this mechanism holds for neural networks more generally. In particular, we show that the right singular vectors and values of the weights can be responsible for the majority of within-class variability collapse for DNNs trained in the feature learning regime. As observed in recent work, this singular structure is highly correlated with that of the AGOP.},
  author       = {Beaglehole, Daniel and Súkeník, Peter and Mondelli, Marco and Belkin, Mikhail},
  booktitle    = {38th Annual Conference on Neural Information Processing Systems},
  issn         = {1049-5258},
  location     = {Vancouver, Canada},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{Average gradient outer product as a mechanism for deep neural collapse}},
  volume       = {37},
  year         = {2024},
}

@inproceedings{18891,
  abstract     = {Deep neural networks (DNNs) exhibit a surprising structure in their final layer
known as neural collapse (NC), and a growing body of works has currently investigated the propagation of neural collapse to earlier layers of DNNs – a phenomenon
called deep neural collapse (DNC). However, existing theoretical results are restricted to special cases: linear models, only two layers or binary classification.
In contrast, we focus on non-linear models of arbitrary depth in multi-class classification and reveal a surprising qualitative shift. As soon as we go beyond two
layers or two classes, DNC stops being optimal for the deep unconstrained features
model (DUFM) – the standard theoretical framework for the analysis of collapse.
The main culprit is a low-rank bias of multi-layer regularization schemes: this bias
leads to optimal solutions of even lower rank than the neural collapse. We support
our theoretical findings with experiments on both DUFM and real data, which show
the emergence of the low-rank structure in the solution found by gradient descent.},
  author       = {Súkeník, Peter and Lampert, Christoph and Mondelli, Marco},
  booktitle    = {38th Annual Conference on Neural Information Processing Systems},
  location     = {Vancouver, Canada},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{Neural collapse versus low-rank bias: Is deep neural collapse really optimal?}},
  volume       = {37},
  year         = {2024},
}

@misc{18895,
  abstract     = {ISTAnt is a new ecological dataset for social immunity and represents the first real-world benchmark for causal inference downstream tasks on high-dimensional observations. It analyzes grooming behavior in the ant Lasius neglectus in groups of three worker ants. The workers for the experiment were obtained from their laboratory stock colony, which had been collected from the field in 2022 in the Botanical Garden Jena, Germany. Ant collection and all experimental work were performed in compliance with international, national and institutional regulations and ethical guidelines. For the experiment, the body surface of one of the three ants was treated with a suspension of either of two microparticle types (diameter ~5 µm) to induce grooming by the two nestmates, which were individually color-coded by application of a dot of blue or orange paint, respectively. The three ants were housed in small plastic containers (diameter 28mm, height 30mm) with moistened, plastered ground and the interior walls covered with PTFE (polytetrafluoroethane) to hamper climbing by the ants. Filming occurred in a temperature- and humidity-controlled room at 23°C within a custom-made filming box with controlled lighting and ventilation conditions. We set up nine ant groups at a time (always containing both treatments) and placed them randomly on positions 1-9 marked on the floor in a 3x3 grid, about 3mm from each other. The experiment was performed on two consecutive days. Videos were acquired using a USB camera (FLIR blackfly S BFS-U3-120S4C, Teledyne FLIR) with a high-performance lens (HP Series 25mm Focal Length, Edmund optics 86-572) in OBS studio 29.0.0 \citep{bailey2017obs} at a framerate of 30 FPS and a resolution of 2500x2500 pixels. From each original video (105x105 mm), we generated nine individual videos .mkv (each ~32x32 mm, 770x770 pixels) by determining exact coordinates per container from one frame in GIMP 2.10.36 and cropping of the videos with FFmpeg 6.1.1. Annotation was performed over two consecutive days by three observers who had not been involved in the experimental setup or recording and were unaware of the treatment assignments to ensure bias-free behavioral annotation. They annotated the behavior of the ants during video observations, using custom-made software that saves the start and end frames of behaviors marked in a .csv file (see 'annotations' folder). In one of the videos, one of the nestmates' legs got inadvertently stuck to its body surface during the color-coding, interfering with its behavior, so the video was discarded. This left 44 videos from 5 independent setups (n=24 of treatment 1 and n=20 of treatment 2) of 10 minutes each for a total of 792 000 annotated frames (see 'video' folder). For each video, we provide the following information: the number of the set to which it belongs (1-5); the number of the position within the set reflecting the position of the ant group under the camera (1-9), for which we also provide ‘coordinates’ in the 3x3 grid (taking values -1/0/1 for both X and Y axis); treatment (1 or 2); the hour of the day when the recording was started (in 24h CEST); experimental day (A or B); the top left coordinate of the cropping square from the original video (CropX/CropY); the person annotating the video (given as A, B, C); the date of annotation (1: first day, 2: second day) and in which order the videos were annotated by each person, both reflecting a possible training effect of the person (see 'experiments_settings.csv' file).},
  author       = {Cadei, Riccardo and Locatello, Francesco and Cremer, Sylvia M and Lindorfer, Lukas and Schmid, Cordelia},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{ISTAnt}},
  doi          = {10.6084/M9.FIGSHARE.26484934.V2},
  year         = {2024},
}

