@article{18064,
  abstract     = { We show that the total number of non-torsion integral points on the elliptic curves ED : y
2 = x3 − D2x, where D ranges over positive squarefree integers less than N, is O(N(log N)
−1/4+ǫ). The proof involves a discriminant-lowering procedure on integral binary quartic forms and an application of Heath-Brown’s method on estimating the average size of the 2-Selmer group of the curves in this family.},
  author       = {Chan, Yik Tung},
  issn         = {1090-2082},
  journal      = {Advances in Mathematics},
  number       = {11},
  publisher    = {Elsevier},
  title        = {{The average number of integral points on the congruent number curves}},
  doi          = {10.1016/j.aim.2024.109946},
  volume       = {457},
  year         = {2024},
}

@article{18173,
  abstract     = {Using a two-dimensional version of the delta method, we establish an asymptotic formula for the number of rational points of bounded height on non-singular complete intersections of cubic and quadric hypersurfaces of dimension at least 23 over Fq(t), provided char (Fq)>3. Under the same hypotheses, we also verify weak approximation.},
  author       = {Glas, Jakob},
  issn         = {1469-7750},
  journal      = {Journal of the London Mathematical Society},
  number       = {4},
  publisher    = {London Mathematical Society},
  title        = {{Rational points on complete intersections of cubic and quadric hypersurfaces over Fq(t)}},
  doi          = {10.1112/jlms.12991},
  volume       = {110},
  year         = {2024},
}

@unpublished{18295,
  abstract     = {By developing a suitable version of the circle method, we show that the space of degree e rational curves on a smooth hypersurface of degree d has only canonical singularities provided its dimension is sufficiently large with respect to e and d.},
  author       = {Glas, Jakob},
  booktitle    = {arXiv},
  title        = {{Canonical singularities on moduli spaces of rational curves via the  circle method}},
  doi          = {10.48550/arXiv.2405.16648},
  year         = {2024},
}

@unpublished{19013,
  abstract     = {We study the singularities of the moduli space of degree e maps from smooth genus g curves to an arbitrary smooth hypersurface of low degree. For e large compared to g, we show that these moduli spaces have at worst terminal singularities. Our main approach is to study the jet schemes of these moduli spaces by developing a suitable form of the circle method.},
  author       = {Glas, Jakob and Hase-Liu, Matthew },
  booktitle    = {arXiv},
  title        = {{Terminal singularities of the moduli space of curves on low degree hypersurfaces and the circle method}},
  doi          = {10.48550/arXiv.2412.14923},
  year         = {2024},
}

@article{14687,
  abstract     = {The short history of research on Li-O2 batteries has seen a remarkable number of mechanistic U-turns over the years. From the initial use of carbonate electrolytes, that were then found to be entirely unsuitable, to the belief that (su)peroxide was solely responsible for degradation, before the more reactive singlet oxygen was found to form, to the hypothesis that capacity depends on a competing surface/solution mechanism before a practically exclusive solution mechanism was identified. Herein, we argue for an ever-fresh look at the reported data without bias towards supposedly established explanations. We explain how the latest findings on rate and capacity limits, as well as the origin of side reactions, are connected via the disproportionation (DISP) step in the (dis)charge mechanism. Therefrom, directions emerge for the design of electrolytes and mediators on how to suppress side reactions and to enable high rate and high reversible capacity.},
  author       = {Jethwa, Rajesh B and Mondal, Soumyadip and Pant, Bhargavi and Freunberger, Stefan Alexander},
  issn         = {1521-3773},
  journal      = {Angewandte Chemie International Edition},
  keywords     = {General Chemistry, Catalysis},
  number       = {28},
  publisher    = {Wiley},
  title        = {{To DISP or not? The far‐reaching reaction mechanisms underpinning Lithium‐air batteries}},
  doi          = {10.1002/anie.202316476},
  volume       = {63},
  year         = {2024},
}

@article{13044,
  abstract     = {Singlet oxygen (1O2) formation is now recognised as a key aspect of non-aqueous oxygen redox chemistry. For identifying 1O2, chemical trapping via 9,10-dimethylanthracene (DMA) to form the endoperoxide (DMA-O2) has become the mainstay method due to its sensitivity, selectivity, and ease of use. While DMA has been shown to be selective for 1O2, rather than forming DMA-O2 with a wide variety of potentially reactive O-containing species, false positives might hypothetically be obtained in the presence of previously overlooked species. Here, we first give unequivocal direct spectroscopic proof by the 1O2-specific near infrared (NIR) emission at 1270 nm for the previously proposed 1O2 formation pathways, which centre around superoxide disproportionation. We then show that peroxocarbonates, common intermediates in metal-O2 and metal carbonate electrochemistry, do not produce false-positive DMA-O2. Moreover, we identify a previously unreported 1O2-forming pathway through the reaction of CO2 with superoxide. Overall, we give unequivocal proof for 1O2 formation in non-aqueous oxygen redox and show that chemical trapping with DMA is a reliable method to assess 1O2 formation.},
  author       = {Mondal, Soumyadip and Jethwa, Rajesh B and Pant, Bhargavi and Hauschild, Robert and Freunberger, Stefan Alexander},
  issn         = {1364-5498},
  journal      = {Faraday Discussions},
  keywords     = {Physical and Theoretical Chemistry},
  pages        = {175--189},
  publisher    = {Royal Society of Chemistry},
  title        = {{Singlet oxygen in non-aqueous oxygen redox: Direct spectroscopic evidence for formation pathways and reliability of chemical probes}},
  doi          = {10.1039/d3fd00088e},
  volume       = {248},
  year         = {2024},
}

@unpublished{20071,
  abstract     = {Farkas established that a system of linear inequalities has a solution if and only if we cannot obtain a contradiction by taking a linear combination of the inequalities. We state and formally prove several Farkas-like theorems over linearly ordered fields in Lean 4. Furthermore, we extend duality theory to the case when some coefficients are allowed to take "infinite values".},
  author       = {Dvorak, Martin and Kolmogorov, Vladimir},
  booktitle    = {arXiv},
  keywords     = {Farkas lemma, linear programming, extended reals, calculus of inductive constructions},
  title        = {{Duality theory in linear optimization and its extensions -- formally  verified}},
  doi          = {10.48550/arXiv.2409.08119},
  year         = {2024},
}

@article{18307,
  abstract     = {Vaccination is the most effective tool to control infectious diseases. However, the evolution of vaccine resistance, exemplified by vaccine resistance in SARS-CoV-2, remains a concern. Here, we model complex vaccination strategies against a pathogen with multiple epitopes—molecules targeted by the vaccine. We found that a vaccine targeting one epitope was ineffective in preventing vaccine escape. Vaccine resistance in highly infectious pathogens was prevented by the full-epitope vaccine, that is, one targeting all available epitopes, but only when the rate of pathogen evolution was low. Strikingly, a bet-hedging strategy of random administration of vaccines targeting different epitopes was the most effective in preventing vaccine resistance in pathogens with the low rate of infection and high rate of evolution. Thus, complex vaccination strategies, when biologically feasible, may be preferable to the currently used single-vaccine approaches for long-term control of disease outbreaks, especially when applied to livestock with near 100% vaccination rates.},
  author       = {Rella, Simon and Kulikova, Yuliya A. and Minnegalieva, Aygul and Kondrashov, Fyodor},
  issn         = {1558-5646},
  journal      = {Evolution: International journal of organic evolution},
  number       = {10},
  pages        = {1722--1738},
  publisher    = {Oxford University Press},
  title        = {{Complex vaccination strategies prevent the emergence of vaccine resistance}},
  doi          = {10.1093/evolut/qpae106},
  volume       = {78},
  year         = {2024},
}

@article{15322,
  abstract     = {The tendency of materials to order in triboelectric series has prompted suggestions that contact electrification might have a single, unified underlying description. However, the possibility of “triboelectric cycles,” i.e., series that loop back onto themselves, is seemingly at odds with such a coherent description. In this work, we propose that if multiple charge carrying species are at play, both triboelectric series and cycles are possible. We show how series arise naturally if only a single charge carrier species is involved and if the driving mechanism is approach toward thermodynamic equilibrium, and simultaneously, that cycles are forbidden under such conditions. Suspecting multiple carriers might relax the situation, we affirm this is the case by explicit construction of a cycle involving two carriers, and then extend this to show how more complex cycles emerge. Our work highlights the importance of series and cycles towards determining the underlying mechanism(s) and carrier(s) in contact electrification.},
  author       = {Sobarzo Ponce, Juan Carlos A and Waitukaitis, Scott R},
  issn         = {2470-0053},
  journal      = {Physical Review E},
  number       = {3},
  publisher    = {American Physical Society},
  title        = {{Multiple charge carrier species as a possible cause for triboelectric cycles}},
  doi          = {10.1103/PhysRevE.109.L032108},
  volume       = {109},
  year         = {2024},
}

@inproceedings{15168,
  abstract     = {A linearly ordered (LO) k-colouring of a hypergraph is a colouring of its vertices with colours 1, … , k such that each edge contains a unique maximal colour. Deciding whether an input hypergraph admits LO k-colouring with a fixed number of colours is NP-complete (and in the special case of graphs, LO colouring coincides with the usual graph colouring). Here, we investigate the complexity of approximating the "linearly ordered chromatic number" of a hypergraph. We prove that the following promise problem is NP-complete: Given a 3-uniform hypergraph, distinguish between the case that it is LO 3-colourable, and the case that it is not even LO 4-colourable. We prove this result by a combination of algebraic, topological, and combinatorial methods, building on and extending a topological approach for studying approximate graph colouring introduced by Krokhin, Opršal, Wrochna, and Živný (2023).},
  author       = {Filakovský, Marek and Nakajima, Tamio Vesa and Opršal, Jakub and Tasinato, Gianluca and Wagner, Uli},
  booktitle    = {41st International Symposium on Theoretical Aspects of Computer Science},
  isbn         = {9783959773119},
  issn         = {1868-8969},
  location     = {Clermont-Ferrand, France},
  publisher    = {Schloss Dagstuhl - Leibniz-Zentrum für Informatik},
  title        = {{Hardness of linearly ordered 4-colouring of 3-colourable 3-uniform hypergraphs}},
  doi          = {10.4230/LIPIcs.STACS.2024.34},
  volume       = {289},
  year         = {2024},
}

@article{18266,
  abstract     = {Matrix games are the most basic model in game theory, and yet robustness with respect to small perturbations of the matrix entries is not fully understood. In this paper, we introduce value positivity and uniform value positivity, two properties that refine the notion of optimality in the context of polynomially perturbed matrix games. The first concept captures how the value depends on the perturbation parameter, and the second consists of the existence of a fixed strategy that guarantees the value of the unperturbed matrix game for every sufficiently small positive parameter. We provide polynomial-time algorithms to check whether a polynomially perturbed matrix game satisfies these properties. We further provide the functional form for a parameterized optimal strategy and the value function. Finally, we translate our results to linear programming and stochastic games, where value positivity is related to the existence of robust solutions.},
  author       = {Chatterjee, Krishnendu and Oliu-Barton, Miquel and Saona Urmeneta, Raimundo J},
  issn         = {1526-5471},
  journal      = {Mathematics of Operations Research},
  number       = {4},
  pages        = {2433--3282},
  publisher    = {Institute for Operations Research and the Management Sciences},
  title        = {{Value-positivity for matrix games}},
  doi          = {10.1287/moor.2022.0332},
  volume       = {50},
  year         = {2024},
}

@unpublished{19545,
  abstract     = {We prove the Eigenstate Thermalisation Hypothesis for Wigner matrices
uniformly in the entire spectrum, in particular near the spectral edges, with a
bound on the fluctuation that is optimal for any observable. This complements
earlier works of Cipolloni et. al. (Comm. Math. Phys. 388, 2021; Forum Math.,
Sigma 10, 2022) and Benigni et. al. (Comm. Math. Phys. 391, 2022; arXiv:
2303.11142) that were restricted either to the bulk of the spectrum or to
special observables. As a main ingredient, we prove a new multi-resolvent local
law that optimally accounts for the edge scaling.},
  author       = {Cipolloni, Giorgio and Erdös, László and Henheik, Sven Joscha},
  booktitle    = {arXiv},
  title        = {{Eigenstate thermalisation at the edge for Wigner matrices}},
  doi          = {10.48550/arXiv.2309.05488},
  year         = {2024},
}

@unpublished{19550,
  abstract     = {We introduce a multi-band BCS free energy functional and prove that for a
multi-band superconductor the effect of inter-band coupling can only increase
the critical temperature, irrespective of its attractive or repulsive nature
and its strength. Further, for weak coupling and weaker inter-band coupling, we
prove that the dependence of the increase in critical temperature on the
inter-band coupling is (1) linear, if there are two or more equally strongly
superconducting bands, or (2) quadratic, if there is only one dominating band.},
  author       = {Henheik, Sven Joscha and Langmann, Edwin and Lauritsen, Asbjørn Bækgaard},
  booktitle    = {arXiv},
  title        = {{Multi-band superconductors have enhanced critical temperatures}},
  doi          = {10.48550/arXiv.2409.17297},
  year         = {2024},
}

@article{17049,
  abstract     = {We consider large non-Hermitian NxN matrices with an additive independent, identically distributed (i.i.d.) noise for each matrix elements. We show that already a small noise of variance 1/N completely thermalises the bulk singular vectors, in particular they satisfy the strong form of Quantum Unique Ergodicity (QUE) with an optimal speed of convergence. In physics terms, we thus extend the Eigenstate Thermalisation Hypothesis, formulated originally by Deutsch [34] and proven for Wigner matrices in [23], to arbitrary non-Hermitian matrices with an i.i.d. noise. As a consequence we obtain an optimal lower bound on the diagonal overlaps of the corresponding non-Hermitian eigenvectors. This quantity, also known as the (square of the) eigenvalue condition number measuring the sensitivity of the eigenvalue to small perturbations, has notoriously escaped rigorous treatment beyond the explicitly computable Ginibre ensemble apart from the very recent upper bounds given in [7] and [45]. As a key tool, we develop a new systematic decomposition of general observables in random matrix theory that governs the size of products of resolvents with deterministic matrices in between.},
  author       = {Cipolloni, Giorgio and Erdös, László and Henheik, Sven Joscha and Schröder, Dominik J},
  issn         = {1096-0783},
  journal      = {Journal of Functional Analysis},
  number       = {4},
  publisher    = {Elsevier},
  title        = {{Optimal lower bound on eigenvector overlaps for non-Hermitian random matrices}},
  doi          = {10.1016/j.jfa.2024.110495},
  volume       = {287},
  year         = {2024},
}

@article{14542,
  abstract     = {It is a remarkable property of BCS theory that the ratio of the energy gap at zero temperature Ξ
 and the critical temperature Tc is (approximately) given by a universal constant, independent of the microscopic details of the fermionic interaction. This universality has rigorously been proven quite recently in three spatial dimensions and three different limiting regimes: weak coupling, low density and high density. The goal of this short note is to extend the universal behavior to lower dimensions d=1,2 and give an exemplary proof in the weak coupling limit.},
  author       = {Henheik, Sven Joscha and Lauritsen, Asbjørn Bækgaard and Roos, Barbara},
  issn         = {1793-6659},
  journal      = {Reviews in Mathematical Physics},
  number       = {9},
  publisher    = {World Scientific Publishing},
  title        = {{Universality in low-dimensional BCS theory}},
  doi          = {10.1142/s0129055x2360005x},
  volume       = {36},
  year         = {2024},
}

@article{18656,
  abstract     = {We consider the time evolution of the out-of-time-ordered correlator (OTOC) of two general observables 
 and 
 in a mean field chaotic quantum system described by a random Wigner matrix as its Hamiltonian. We rigorously identify three time regimes separated by the physically relevant scrambling and relaxation times. The main feature of our analysis is that we express the error terms in the optimal Schatten (tracial) norms of the observables, allowing us to track the exact dependence of the errors on their rank. In particular, for significantly overlapping observables with low rank the OTOC is shown to exhibit a significant local maximum at the scrambling time, a feature that may not have been noticed in the physics literature before. Our main tool is a novel multi-resolvent local law with Schatten norms that unifies and improves previous local laws involving either the much cruder operator norm (cf. [10]) or the Hilbert-Schmidt norm (cf. [11]).},
  author       = {Cipolloni, Giorgio and Erdös, László and Henheik, Sven Joscha},
  issn         = {1095-0753},
  journal      = {Advances in Theoretical and Mathematical Physics},
  number       = {6},
  pages        = {2025--2083},
  publisher    = {International Press of Boston},
  title        = {{Out-of-time-ordered correlators for Wigner matrices}},
  doi          = {10.4310/ATMP.241031013250},
  volume       = {28},
  year         = {2024},
}

@unpublished{19547,
  abstract     = {For correlated real symmetric or complex Hermitian random matrices, we prove
that the local eigenvalue statistics at any cusp singularity are universal.
Since the density of states typically exhibits only square root edge or cubic
root cusp singularities, our result completes the proof of the
Wigner-Dyson-Mehta universality conjecture in all spectral regimes for a very
general class of random matrices. Previously only the bulk and the edge
universality were established in this generality [arXiv:1804.07744], while cusp
universality was proven only for Wigner-type matrices with independent entries
[arXiv:1809.03971, arXiv:1811.04055]. As our main technical input, we prove an
optimal local law at the cusp using the Zigzag strategy, a recursive tandem of
the characteristic flow method and a Green function comparison argument.
Moreover, our proof of the optimal local law holds uniformly in the spectrum,
thus also re-establishing universality of the local eigenvalue statistics in
the previously studied bulk [arXiv:1705.10661] and edge [arXiv:1804.07744]
regimes.},
  author       = {Erdös, László and Henheik, Sven Joscha and Riabov, Volodymyr},
  booktitle    = {arXiv},
  title        = {{Cusp universality for correlated random matrices}},
  doi          = {10.48550/arXiv.2410.06813},
  year         = {2024},
}

@unpublished{19551,
  abstract     = {We introduce a notion of a \emph{local gap} for interacting many-body quantum lattice systems and prove the validity of response theory and Kubo's formula for localized perturbations in such settings.
On a high level, our result shows that the usual spectral gap condition, concerning the system as a whole, is not a necessary condition for understanding local properties of the system.
More precisely, we say that an equilibrium state ρ0 of a Hamiltonian H0 is locally gapped in Λgap⊂Λ, whenever the Liouvillian −i[H0,⋅] is almost invertible on local observables supported in Λgap when tested in ρ0.
To put this into context, we provide other alternative notions of a local gap and discuss their relations.
The validity of response theory is based on the construction of \emph{non-equilibrium almost stationary states} (NEASSs).
By controlling locality properties of the NEASS construction, we show that response theory holds to any order, whenever the perturbation \(\epsilon V\) acts in a region which is further than |logϵ| away from the non-gapped region Λ∖Λgap.},
  author       = {Henheik, Sven Joscha and Wessel, Tom},
  booktitle    = {arXiv},
  title        = {{Response theory for locally gapped systems}},
  doi          = {10.48550/arXiv.2410.10809},
  year         = {2024},
}

@phdthesis{17485,
  abstract     = {Large language models (LLMs) have made tremendous progress in the past few years, from being able to generate coherent text to matching or surpassing humans in a wide variety of creative, knowledge or reasoning tasks. Much of this can be attributed to massively increased scale, both in the size of the model as well as the amount of training data, from 100s of millions to 100s of billions, or even trillions. This trend is expected to continue, which, although exciting, also raises major practical concerns. Already today's 100+ billion parameter LLMs require top-of-the-line hardware just to run. Hence, it is clear that sustaining these developments will require significant efficiency advances.

Historically, one of the most practical ways of improving model efficiency has been compression, especially in the form of sparsity or quantization. While this has been studied extensively in the past, existing accurate methods are all designed for models around 100 million parameters; scaling them up to ones literally 1000x larger is highly challenging. In this thesis, we introduce a new unified sparsification and quantization approach OBC, which through additional algorithmic enhancements leads to GPTQ and SparseGPT, the first techniques fast and accurate enough to compress 100+ billion parameter models to 4- or even 3-bit precision and 50% weight-sparsity, respectively. Additionally, we show how weight-only quantizion does not just bring space savings but also up to 4.5x faster generation speed, via custom GPU kernels.

In fact, we show for the first time that it is possible to develop an FP16 times INT4 mixed-precision matrix multiplication kernel, called Marlin, which comes close to simultaneously maximizing both memory and compute utilization, making weight-only quantization highly practical even for multi-user serving. Further, we demonstrate that GPTQ can be scaled to widely overparametrized trillion-parameter models, where extreme sub-1-bit compression rates can be achieved without any inference slow-down, by co-designing a bespoke entropy coding scheme together with an efficient kernel.

Finally, we also study compression from the perspective of someone with access to massive amounts of compute resources for training large models completely from scratch. Here the key questions evolve around the joint scaling behavior between compression, model size, and amount of training data used. Based on extensive experimental results for both vision and text models, we introduce the first scaling law which accurately captures the relationship between weight-sparsity, number of non-zero weights and data. This further allows us to characterize the optimal sparsity, which we find to increase the longer a fixed cost model is being trained.

Overall, this thesis presents contributions to three different angles of large model efficiency: affordable but accurate algorithms, highly efficient systems implementations, and fundamental scaling laws for compressed training.},
  author       = {Frantar, Elias},
  issn         = {2663-337X},
  pages        = {129},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Compressing large neural networks: Algorithms, systems and scaling laws}},
  doi          = {10.15479/at:ista:17485},
  year         = {2024},
}

@inproceedings{18062,
  abstract     = {We explore the impact of parameter sparsity on the scaling behavior of Transformers trained on massive datasets (i.e., "foundation models"), in both vision and language domains. In this setting, we identify the first scaling law describing the relationship between weight sparsity, number of non-zero parameters, and amount of training data, which we validate empirically across model and data scales; on ViT/JFT-4B and T5/C4. These results allow us to characterize the "optimal sparsity", the sparsity level which yields the best performance for a given effective model size and training budget. For a fixed number of non-zero parameters, we identify that the optimal sparsity increases with the amount of data used for training. We also extend our study to different sparsity structures (such as the hardware-friendly n:m pattern) and strategies (such as starting from a pretrained dense model). Our findings shed light on the power and limitations of weight sparsity across various parameter and computational settings, offering both theoretical understanding and practical implications for leveraging sparsity towards computational efficiency improvements. We provide pruning and scaling law fitting code at: github.com/google-research/jaxpruner/tree/main/jaxpruner/projects/bigsparse.},
  author       = {Frantar, Elias and Ruiz, Carlos Riquelme and Houlsby, Neil and Alistarh, Dan-Adrian and Evci, Utku},
  booktitle    = {The Twelfth International Conference on Learning Representations},
  location     = {Vienna, Austria},
  title        = {{Scaling laws for sparsely-connected foundation models}},
  year         = {2024},
}

