@article{21343,
  abstract     = {The large sieve is used to estimate the density of quadratic polynomials Q ∈ Z[x],
such that there exists an odd degree polynomial defined over Z which has resultant ±1 with Q.
Given a monic polynomial R ∈ Z[x] of odd degree, this is used to show that for almost all
quadratic polynomials Q ∈ Z[x], there exists a prime p such that Q and R share a common
root in Fp. Using recent work of Landesman, an application to the average size of the odd part
of the class group of quadratic number fields is also given},
  author       = {Browning, Timothy D and Chan, Yik Tung},
  issn         = {2270-518X},
  journal      = {Journal de l'Ecole Polytechnique - Mathematiques},
  pages        = {1677--1691},
  publisher    = {Ecole Polytechnique},
  title        = {{Solubility of a resultant equation and applications}},
  doi          = {10.5802/jep.320},
  volume       = {12},
  year         = {2025},
}

@article{15033,
  abstract     = {The GNOM (GN) Guanine nucleotide Exchange Factor for ARF small GTPases (ARF-GEF) is among the best studied trafficking regulators in plants, playing crucial and unique developmental roles in patterning and polarity. The current models place GN at the Golgi apparatus (GA), where it mediates secretion/recycling, and at the plasma membrane (PM) presumably contributing to clathrin-mediated endocytosis (CME). The mechanistic basis of the developmental function of GN, distinct from the other ARF-GEFs including its closest homologue GNOM-LIKE1 (GNL1), remains elusive. Insights from this study largely extend the current notions of GN function. We show that GN, but not GNL1, localizes to the cell periphery at long-lived structures distinct from clathrin-coated pits, while CME and secretion proceed normally in <jats:italic>gn</jats:italic> knockouts. The functional GN mutant variant GN<jats:sup>fewerroots</jats:sup>, absent from the GA, suggests that the cell periphery is the major site of GN action responsible for its developmental function. Following inhibition by Brefeldin A, GN, but not GNL1, relocates to the PM likely on exocytic vesicles, suggesting selective molecular associations en route to the cell periphery. A study of GN-GNL1 chimeric ARF-GEFs indicates that all GN domains contribute to the specific GN function in a partially redundant manner. Together, this study offers significant steps toward the elucidation of the mechanism underlying unique cellular and development functions of GNOM.},
  author       = {Adamowski, Maciek and Matijevic, Ivana and Friml, Jiří},
  issn         = {2050-084X},
  journal      = {eLife},
  keywords     = {General Immunology and Microbiology, General Biochemistry, Genetics and Molecular Biology, General Medicine, General Neuroscience},
  publisher    = {eLife Sciences Publications},
  title        = {{Developmental patterning function of GNOM ARF-GEF mediated from the cell periphery}},
  doi          = {10.7554/elife.68993},
  volume       = {13},
  year         = {2024},
}

@misc{15385,
  abstract     = {Relevant information about the data can be found in the 'Readme_Data.txt' file. 
A previous version of the publication can be found on BioRxiv: https://www.biorxiv.org/content/10.1101/2022.10.11.511691v4
and published in Plos Biology (2024)},
  author       = {Burnett, Laura and Koppensteiner, Peter and Symonova, Olga and Masson, Tomas and Vega Zuniga, Tomas A and Contreras, Ximena and Rülicke, Thomas and Shigemoto, Ryuichi and Novarino, Gaia and Jösch, Maximilian A},
  keywords     = {ASD, periaqueductal gray, perception, behavior, potassium channels},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Shared behavioural impairments in visual perception and place avoidance across different autism models are driven by periaqueductal grey hypoexcitability in Setd5 haploinsufficient mice}},
  doi          = {10.15479/AT:ISTA:15385},
  year         = {2024},
}

@misc{18498,
  abstract     = {Scripts and data used in the research study Predicting rapid adaptation in time from adaptation in space: a 30-year field experiment in marine snails. https://doi.org/10.1101/2023.09.27.559715},
  author       = {Garcia Castillo, Diego Fernando and Barton, Nicholas H and Faria, Rui and Larsson, Jenny and Stankowski, Sean and Butlin, Roger and Johannesson, Kerstin and Westram, Anja M},
  publisher    = {Zenodo},
  title        = {{Data and code for: Predicting rapid adaptation in time from adaptation in space: a 30-year field experiment in marine snails}},
  doi          = {10.5281/ZENODO.12159343},
  year         = {2024},
}

@article{18856,
  abstract     = {This research is aimed to solve the tweet/user geolocation prediction task and provide a flexible methodology for the geo-tagging of textual big data. The suggested approach implements neural networks for natural language processing (NLP) to estimate the location as coordinate pairs (longitude, latitude) and two-dimensional Gaussian Mixture Models (GMMs). The scope of proposed models has been finetuned on a Twitter dataset using pretrained Bidirectional Encoder Representations from Transformers (BERT) as base models. Performance metrics show a median error of fewer than 30 km on a worldwide-level, and fewer than 15 km on the US-level datasets for the models trained and evaluated on text features of tweets' content and metadata context. Our source code and data are available at https://github.com/K4TEL/geo-twitter.git.},
  author       = {Lutsai, Kateryna and Lampert, Christoph},
  issn         = {1948-660X},
  journal      = {Journal of Spatial Information Science},
  number       = {29},
  pages        = {69--99},
  publisher    = {University of Maine},
  title        = {{Predicting the geolocation of tweets using transformer models on customized data}},
  doi          = {10.5311/JOSIS.2024.29.295},
  year         = {2024},
}

@article{18934,
  abstract     = {The assembly of biomolecular condensate in eukaryotic cells and the accumulation of amyloid deposits in neurons are processes involving the nucleation and growth (NAG) of new protein phases. To therapeutically target protein phase separation, drug candidates are tested in in vitro assays that monitor the increase in the mass or size of the new phase. Limited mechanistic insight is, however, provided if empirical or untestable kinetic models are fitted to these progress curves. Here we present the web server NAGPKin that quantifies NAG rates using mass-based or size-based progress curves as the input data. A report is generated containing the fitted NAG parameters and elucidating the phase separation mechanisms at play. The NAG parameters can be used to predict particle size distributions of, for example, protein droplets formed by liquid-liquid phase separation (LLPS) or amyloid fibrils formed by protein aggregation. Because minimal intervention is required from the user, NAGPKin is a good platform for standardized reporting of LLPS and protein self-assembly data. NAGPKin is useful for drug discovery as well as for fundamental studies on protein phase separation. NAGPKin is freely available (no login required) at https://nagpkin.i3s.up.pt .},
  author       = {Sárkány, Zsuzsa and Figueiredo, Francisco and Macedo-Ribeiro, Sandra and Martins, Pedro M.},
  issn         = {1939-4586},
  journal      = {Molecular Biology of the Cell},
  number       = {3},
  publisher    = {American Society for Cell Biology},
  title        = {{NAGPKin: Nucleation-and-growth parameters from the kinetics of protein phase separation}},
  doi          = {10.1091/mbc.e23-07-0289},
  volume       = {35},
  year         = {2024},
}

@inproceedings{18956,
  abstract     = {Group Activity Recognition (GAR) aims to detect the activity performed by multiple actors in a scene. Prior works model the spatio-temporal features based on the RGB, optical flow or keypoint data types. On the contrary, our hypothesis is that by only using the RGB data without temporality, the performance can be maintained with a negligible loss in accuracy. To that end, we propose a novel GAR technique for volleyball videos, DECOMPL, which consists of two complementary branches. In the visual branch, it extracts the features using attention pooling. In the coordinate branch, it considers the configuration of the players and extracts the spatial information from the box coordinates. Moreover, we analyzed the Volleyball dataset that the recent literature is mostly based on, and systematically reannotated it to emphasize the group concept. Experimental results demonstrated the effectiveness of the proposed model DECOMPL, which delivered the best/second best GAR performance with the reannotations/original annotations among the comparable state-of-the-art methods. Code and new annotations are available at GitHub: https://github.com/berkerdemirel/decompl},
  author       = {Demirel, Berker and Ozkan, Huseyin},
  booktitle    = {2024 IEEE International Conference on Image Processing},
  issn         = {2381-8549},
  location     = {Abu Dhabi, United Arab Emirates},
  pages        = {977--983},
  publisher    = {IEEE},
  title        = {{Decompl: Decompositional learning with attention pooling for group activity recognition from a single volleyball image}},
  doi          = {10.1109/icip51287.2024.10647499},
  year         = {2024},
}

@inproceedings{18964,
  abstract     = {Object-centric learning (OCL) extracts the representation of objects with slots, offering an exceptional blend of flexibility and interpretability for abstracting low-level perceptual features. A widely adopted method within OCL is slot attention, which utilizes attention mechanisms to iteratively refine slot representations. However, a major draw-back of most object-centric models, including slot attention, is their reliance on predefining the number of slots. This not only necessitates prior knowledge of the dataset but also overlooks the inherent variability in the number of objects present in each instance. To overcome this fundamental limitation, we present a novel complexity-aware object auto-encoder framework. Within this framework, we introduce an adaptive slot attention (AdaSlot) mecha-nism that dynamically determines the optimal number of slots based on the content of the data. This is achieved by proposing a discrete slot sampling module that is responsible for selecting an appropriate number of slots from a candidate list. Furthermore, we introduce a masked slot decoder that suppresses unselected slots during the decoding process. Our framework, tested extensively on object discovery tasks with various datasets, shows performance matching or exceeding top fixed-slot models. Moreover, our analysis substantiates that our method exhibits the capability to dynamically adapt the slot number according to each instance's complexity, offering the potential for further exploration in slot attention research. Project will be available at https://kfan21.github.io/AdaSlot/},
  author       = {Fan, Ke and Bai, Zechen and Xiao, Tianjun and He, Tong and Horn, Max and Fu, Yanwei and Locatello, Francesco and Zhang, Zheng},
  booktitle    = {2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition},
  location     = {Seattle, WA, United States},
  publisher    = {IEEE},
  title        = {{Adaptive slot attention: Object discovery with dynamic slot number}},
  doi          = {10.1109/cvpr52733.2024.02176},
  year         = {2024},
}

@inproceedings{18971,
  abstract     = {Models prone to spurious correlations in training data often produce brittle predictions and introduce unintended biases. Addressing this challenge typically involves methods relying on prior knowledge and group annotation to remove spurious correlations, which may not be readily available in many applications. In this paper, we establish a novel connection between unsupervised object-centric learning and mitigation of spurious correlations. Instead of directly inferring subgroups with varying correlations with labels, our approach focuses on discovering concepts: discrete ideas that are shared across input samples. Leveraging existing object-centric representation learning, we introduce CoBalT: a concept balancing technique that effectively mitigates spurious correlations without requiring human labeling of subgroups. Evaluation across the benchmark datasets for sub-population shifts demonstrate superior or competitive performance compared state-of-the-art baselines, without the need for group annotation. Code is available at https://github.com/rarefin/CoBalT},
  author       = {Arefin, Rifat and Zhang, Yan and Baratin, Aristide and Locatello, Francesco and Rish, Irina and Liu, Dianbo and Kawaguchi, Kenji},
  booktitle    = {Proceedings of the 41st International Conference on Machine Learning},
  issn         = {2640-3498},
  location     = {Vienna, Austria},
  pages        = {1672--1688},
  publisher    = {ML Research Press},
  title        = {{Unsupervised concept discovery mitigates spurious correlations}},
  volume       = {235},
  year         = {2024},
}

@inproceedings{18996,
  abstract     = {We consider the linear causal representation learning setting where we observe a linear mixing of d unknown latent factors, which follow a linear structural causal model. Recent work has shown that it is possible to recover the latent factors as well as the underlying structural causal model over them, up to permutation and scaling, provided that we have at least d environments, each of which corresponds to perfect interventions on a single latent node (factor). After this powerful result, a key open problem faced by the community has been to relax these conditions: allow for coarser than perfect single-node interventions, and allow for fewer than d of them, since the number of latent factors d could be very large. In this work, we consider precisely such a setting, where we allow a smaller than d number of environments, and also allow for very coarse interventions that can very coarsely \textit{change the entire causal graph over the latent factors}. On the flip side, we relax what we wish to extract to simply the \textit{list of nodes that have shifted between one or more environments}. We provide a surprising identifiability result that it is indeed possible, under some very mild standard assumptions, to identify the set of shifted nodes. Our identifiability proof moreover is a constructive one: we explicitly provide necessary and sufficient conditions for a node to be a shifted node, and show that we can check these conditions given observed data. Our algorithm lends itself very naturally to the sample setting where instead of just interventional distributions, we are provided datasets of samples from each of these distributions. We corroborate our results on both synthetic experiments as well as an interesting psychometric dataset. The code can be found at https://github.com/TianyuCodings/iLCS.},
  author       = {Chen, Tianyu and Bello, Kevin and Locatello, Francesco and Aragam, Bryon and Ravikumar, Pradeep Kumar},
  booktitle    = {38th Conference on Neural Information Processing Systems},
  issn         = {1049-5258},
  location     = {Vancouver, Canada},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{Identifying general mechanism shifts in linear causal representations}},
  volume       = {37},
  year         = {2024},
}

@inproceedings{19005,
  abstract     = {Causal representation learning promises to extend causal models to hidden causal
variables from raw entangled measurements. However, most progress has focused
on proving identifiability results in different settings, and we are not aware of any
successful real-world application. At the same time, the field of dynamical systems
benefited from deep learning and scaled to countless applications but does not allow
parameter identification. In this paper, we draw a clear connection between the two
and their key assumptions, allowing us to apply identifiable methods developed
in causal representation learning to dynamical systems. At the same time, we can
leverage scalable differentiable solvers developed for differential equations to build
models that are both identifiable and practical. Overall, we learn explicitly controllable models that isolate the trajectory-specific parameters for further downstream
tasks such as out-of-distribution classification or treatment effect estimation. We
experiment with a wind simulator with partially known factors of variation. We
also apply the resulting model to real-world climate data and successfully answer
downstream causal questions in line with existing literature on climate change.
Code is available at https://github.com/CausalLearningAI/crl-dynamical-systems.},
  author       = {Yao, Dingling and Muller, Caroline J and Locatello, Francesco},
  booktitle    = {38th Conference on Neural Information Processing Systems},
  location     = {Vancouver, Canada},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{Marrying causal representation learning with dynamical systems for science}},
  volume       = {37},
  year         = {2024},
}

@article{19486,
  abstract     = {Consider the family of elliptic curves En:y2=x3+n2, where n varies over positive cubefree integers. There is a rational 3-isogeny ϕ from En to E^n:y2=x3−27n2 and a dual isogeny ϕ^:E^n→En. We show that for almost all n, the rank of Selϕ(En) is 0, and the rank of Selϕ^(E^n) is determined by the number of prime factors of n that are congruent to 2mod3 and the congruence class of nmod9.},
  author       = {Chan, Yik Tung},
  issn         = {1687-0247},
  journal      = {International Mathematics Research Notices},
  number       = {9},
  pages        = {7571--7593},
  publisher    = {Oxford University Press},
  title        = {{The 3-isogeny selmer groups of the elliptic curves y2=x3+n2}},
  doi          = {10.1093/imrn/rnad266},
  volume       = {2024},
  year         = {2024},
}

@inproceedings{19510,
  abstract     = {We propose a new variant of the Adam optimizer [Kingma and Ba, 2014] called
MICROADAM that specifically minimizes memory overheads, while maintaining
theoretical convergence guarantees. We achieve this by compressing the gradient
information before it is fed into the optimizer state, thereby reducing its memory
footprint significantly. We control the resulting compression error via a novel
instance of the classical error feedback mechanism from distributed optimization [Seide et al., 2014, Alistarh et al., 2018, Karimireddy et al., 2019] in which
the error correction information is itself compressed to allow for practical memory
gains. We prove that the resulting approach maintains theoretical convergence
guarantees competitive to those of AMSGrad, while providing good practical performance. Specifically, we show that MICROADAM can be implemented efficiently
on GPUs: on both million-scale (BERT) and billion-scale (LLaMA) models, MICROADAM provides practical convergence competitive to that of the uncompressed
Adam baseline, with lower memory usage and similar running time. Our code is
available at https://github.com/IST-DASLab/MicroAdam.},
  author       = {Modoranu, Ionut-Vlad and Safaryan, Mher and Malinovsky, Grigory and Kurtic, Eldar and Robert, Thomas and Richtárik, Peter and Alistarh, Dan-Adrian},
  booktitle    = {38th Conference on Neural Information Processing Systems},
  issn         = {1049-5258},
  publisher    = {Neural Information Processing Systems Foundation},
  title        = {{MICROADAM: Accurate adaptive optimization with low space overhead and provable convergence}},
  volume       = {37},
  year         = {2024},
}

@article{20060,
  abstract     = {Un fenómeno a menudo asociado con el autismo es un modo atípico de función ejecutiva, cuyas manifestaciones incluyen dificultad para iniciar tareas. En algunos casos, esto va acompañado de sentimientos de inercia y sensaciones que pueden describirse como inquietud y parálisis simultáneas. En consecuencia, la dificultad para iniciar las tareas puede dar lugar a la procrastinación, ya sea simplemente posponiendo el trabajo en la tarea objetivo o realizando otras tareas no relacionadas antes de dedicarse a la tarea objetivo. Curiosamente, sin embargo, también está documentado que, una vez iniciada una tarea, los autistas pueden centrarse en ella intensamente y durante periodos prolongados de tiempo, especialmente cuando les resulta interesante.&#x0D;
Este trabajo utiliza el procesamiento predictivo y la inferencia activa para modelar la relación entre la función ejecutiva, la procrastinación y la hiperfocalización en el autismo. Este modelo integra las causas conocidas y propuestas de los déficits en la función ejecutiva y el papel que desempeña el interés en la regulación de la atención y la motivación. El modelo propone que la procrastinación es el resultado de procesos diferenciales de minimización de errores de predicción, como la ponderación de estímulos sensoriales. Se discuten los vínculos con modelos propuestos previamente, como la coherencia central débil (CCC), y la teoría de los priores altos e inflexibles de los errores de predicción en el autismo (HIPPEA).},
  author       = {Carls-Diamante, Sidney and Laciny, Alice},
  issn         = {1316-693X},
  journal      = {Lógoi. Revista de Filosofía},
  number       = {45},
  pages        = {88--114},
  publisher    = {Universidad Católica Andrés Bello},
  title        = {{Stuck in uncertainty: A predictive processing/ active inference account of procrastination-like behaviour in autism}},
  doi          = {10.62876/lr.vi45.6481},
  year         = {2024},
}

@article{20527,
  abstract     = {Arising from C. Yang et al. Nature Chemistry https://doi.org/10.1038/s41557-023-01212-2 (2023)

In this work Yang et al.1 claim that an enantioselective Michael addition reaction with a barrier of 16 kcal mol−1 occurs at the single-molecule level in frozen solvent by measuring fluctuations in current flowing across graphene-based molecular devices. The article, however, contains major scientific errors that undermine their conclusions. We highlight issues with the fabrication of the devices, a lack of characterization, discrepancies between theory and experiment, unreliable inelastic electron tunnelling spectra (IETS) and a perceived misinterpretation of noise as evidence of reaction.},
  author       = {Venkataraman, Latha and van Ruitenbeek, Jan},
  issn         = {1755-4349},
  journal      = {Nature Chemistry},
  number       = {11},
  pages        = {1767--1769},
  publisher    = {Springer Nature},
  title        = {{Questioning claims of monitoring the Michael addition reaction at the single-molecule level}},
  doi          = {10.1038/s41557-024-01631-9},
  volume       = {16},
  year         = {2024},
}

@book{20615,
  abstract     = {Spin/Pin-structures on vector bundles have long featured prominently in differential geometry, in particular providing part of the foundation for the original proof of the renowned Atiyah–Singer Index Theory. More recently, they have underpinned the symplectic topology foundations of the so-called real sector of the mirror symmetry of string theory.

This semi-expository three-part monograph provides an accessible introduction to Spin- and Pin-structures in general, demonstrates their role in the orientability considerations in symplectic topology, and presents their applications in enumerative geometry.

Part I contains a systematic treatment of Spin/Pin-structures from different topological perspectives and may be suitable for an advanced undergraduate reading seminar. This leads to Part II, which systematically studies orientability problems for the determinants of real Cauchy–Riemann operators on vector bundles. Part III introduces enumerative geometry of curves in complex projective varieties and in symplectic manifolds, demonstrating some applications of the first two parts in the process. Two appendices review the Čech cohomology perspective on fiber bundles and Lie group covering spaces.},
  author       = {Chen, Xujia and Zinger, Aleksey},
  isbn         = {9789811278532},
  publisher    = {World Scientific Publishing},
  title        = {{Spin/Pin-structures and real enumerative geometry}},
  doi          = {10.1142/13476},
  year         = {2024},
}

@inproceedings{17456,
  abstract     = {Data-parallel distributed training of deep neural networks (DNN) has gained very widespread adoption, but can still experience communication bottlenecks. To address this issue, entire families of compression mechanisms have been developed, including quantization, sparsification, and low-rank approximation, some of which are seeing significant practical adoption. Despite this progress, almost all known compression schemes apply compression uniformly across DNN layers, although layers are heterogeneous in terms of parameter count and their impact on model accuracy.In this work, we provide a general framework for adapting the degree of compression across the model's layers dynamically during training, improving the overall compression, while leading to substantial speedups, without sacrificing accuracy. Our framework, called L-GreCo, is based on an adaptive algorithm, which automatically picks the optimal compression parameters for model layers guaranteeing the best compression ratio while satisfying an error constraint. Extensive experiments over image classification and language modeling tasks shows that L-GreCo is effective across all existing families of compression methods, and achieves up to 2.5
×
 training speedup and up to 5
×
 compression improvement over efficient implementations of existing approaches, while recovering full accuracy. Moreover, L-GreCo is complementary to existing adaptive algorithms, improving their compression ratio by 50\% and practical throughput by 66\%. An anonymized implementation is available at https://github.com/LGrCo/L-GreCo.},
  author       = {Markov, Ilia and Alimohammadi, Kaveh and Frantar, Elias and Alistarh, Dan-Adrian},
  booktitle    = {Proceedings of Machine Learning and Systems },
  editor       = {Gibbons, P. and Pekhimenko, G. and De Sa, C.},
  location     = {Athens, Greece},
  publisher    = {Association for Computing Machinery},
  title        = {{L-GreCo: Layerwise-adaptive gradient compression for efficient data-parallel deep learning}},
  volume       = {6},
  year         = {2024},
}

@inproceedings{18114,
  abstract     = {This paper presents Mechanistic Neural Networks, a neural network design for machine learning applications in the sciences. It incorporates a new Mechanistic Block in standard architectures to explicitly learn governing differential equations as representations, revealing the underlying dynamics of data and enhancing interpretability and efficiency in data modeling. Central to our approach is a novel Relaxed Linear Programming Solver (NeuRLP) inspired by a technique that reduces solving linear ODEs to solving linear programs. This integrates well with neural networks and surpasses the limitations of traditional ODE solvers enabling scalable GPU parallel processing. Overall, Mechanistic Neural Networks demonstrate their versatility for scientific machine learning applications, adeptly managing tasks from equation discovery to dynamic systems modeling. We prove their comprehensive capabilities in analyzing and interpreting complex scientific data across various applications, showing significant performance against specialized state-of-the-art methods. Source code is available at https://github.com/alpz/mech-nn.},
  author       = {Pervez, Adeel A and Locatello, Francesco and Gavves, Efstratios},
  booktitle    = {Proceedings of the 41st International Conference on Machine Learning},
  issn         = {2640-3498},
  location     = {Vienna, Austria},
  pages        = {40484--40501},
  publisher    = {ML Research Press},
  title        = {{Mechanistic neural networks for scientific machine learning}},
  volume       = {235},
  year         = {2024},
}

@inproceedings{18117,
  abstract     = {We investigate parameter-efficient fine-tuning (PEFT) methods that can provide good accuracy under limited computational and memory budgets in the context of large language models (LLMs). We present a new PEFT method called Robust Adaptation (RoSA) inspired by robust principal component analysis that jointly trains low-rank
 and highly-sparse components on top of a set of fixed pretrained weights to efficiently approximate the performance of a full-fine-tuning (FFT) solution. Across a series of challenging generative tasks such as grade-school math and SQL query generation, which require fine-tuning for good performance, we show that RoSA outperforms LoRA, pure sparse fine-tuning, and alternative hybrid methods at the same parameter budget, and can even recover the performance of FFT on some tasks. We provide system support for RoSA to complement the training algorithm, specifically in the form of sparse GPU kernels which enable memory- and computationally-efficient training, and show that it is also compatible with low-precision base weights, resulting in the first joint representation combining quantization, low-rank and sparse approximations. Our code is available at https://github.com/IST-DASLab/RoSA.},
  author       = {Nikdan, Mahdi and Tabesh, Soroush and Crncevic, Elvir and Alistarh, Dan-Adrian},
  booktitle    = {Proceedings of the 41st International Conference on Machine Learning},
  issn         = {2640-3498},
  location     = {Vienna, Austria},
  pages        = {38187--38206},
  publisher    = {ML Research Press},
  title        = {{RoSA: Accurate parameter-efficient fine-tuning via robust adaptation}},
  volume       = {235},
  year         = {2024},
}

@article{22197,
  abstract     = {We study the distribution of consecutive sums of two squares
in arithmetic progressions. If {En}n∈N is the sequence of
sums of two squares in increasing order, we show that for
any modulus q and any congruence classes a1, a2, a3 mod q
which are admissible in the sense that there are solutions
to x2 + y2 ≡ ai mod q, there exist infinitely many n with
En+i−1 ≡ ai mod q, for i =1, 2, 3. We also show that for
any r1, r2 ≥ 1, there exist infinitely many n with En+i−1 ≡
a1 mod q for 1 ≤ i ≤ r1 and En+i−1 ≡ a2 mod q for
r1 +1 ≤ i ≤ r1 + r2},
  author       = {Kimmel, Noam and Kuperberg, Vivian Zieve},
  issn         = {0022-314X},
  journal      = {Journal of Number Theory},
  pages        = {135--147},
  publisher    = {Elsevier},
  title        = {{Consecutive runs of sums of two squares}},
  doi          = {10.1016/j.jnt.2024.05.003},
  volume       = {264},
  year         = {2024},
}

