@inproceedings{17378,
  abstract     = {Generative Pre-trained Transformer models, known as GPT or OPT, set themselves apart through breakthrough performance across complex language modelling tasks, but also by their extremely high computational and storage costs. Specifically, due to their massive size, even inference for large, highly-accurate GPT models may require multiple performant GPUs, which limits the usability of such models. While there is emerging work on relieving this pressure via model compression, the applicability and performance of existing compression techniques is limited by the scale and complexity of GPT models. In this paper, we address this challenge, and propose OPTQ, a new one-shot weight quantization method based on approximate second-order information, that is both highly-accurate and highly-efficient. Specifically, OPTQ can quantize GPT models with 175 billion parameters in approximately four GPU hours, reducing the bitwidth down to 3 or 4 bits per weight, with negligible accuracy degradation relative to the uncompressed baseline. Our method more than doubles the compression gains relative to previously-proposed one-shot quantization methods, preserving accuracy, allowing us for the first time to execute an 175 billion-parameter model inside a single GPU for generative inference. Moreover, we also show that our method can still provide reasonable accuracy in the extreme quantization regime, in which weights are quantized to 2-bit or even ternary quantization levels. We show experimentally that these improvements can be leveraged for end-to-end inference speedups over FP16, of around 3.25x when using high-end GPUs (NVIDIA A100) and 4.5x when using more cost-effective ones (NVIDIA A6000). The implementation is available at https://github.com/IST-DASLab/gptq.},
  author       = {Frantar, Elias and Ashkboos, Saleh and Hoefler, Torsten and Alistarh, Dan-Adrian},
  booktitle    = {11th International Conference on Learning Representations },
  location     = {Kigali, Rwanda},
  publisher    = {International Conference on Learning Representations},
  title        = {{OPTQ: Accurate post-training quantization for generative pre-trained transformers}},
  year         = {2023},
}

@inproceedings{14458,
  abstract     = {We show for the first time that large-scale generative pretrained transformer (GPT) family models can be pruned to at least 50% sparsity in one-shot, without any retraining, at minimal loss of accuracy. This is achieved via a new pruning method called SparseGPT, specifically designed to work efficiently and accurately on massive GPT-family models. We can execute SparseGPT on the largest available open-source models, OPT-175B and BLOOM-176B, in under 4.5 hours, and can reach 60% unstructured sparsity with negligible increase in perplexity: remarkably, more than 100 billion weights from these models can be ignored at inference time. SparseGPT generalizes to semi-structured (2:4 and 4:8) patterns, and is compatible with weight quantization approaches. The code is available at: https://github.com/IST-DASLab/sparsegpt.},
  author       = {Frantar, Elias and Alistarh, Dan-Adrian},
  booktitle    = {Proceedings of the 40th International Conference on Machine Learning},
  issn         = {2640-3498},
  location     = {Honolulu, Hawaii, HI, United States},
  pages        = {10323--10337},
  publisher    = {ML Research Press},
  title        = {{SparseGPT: Massive language models can be accurately pruned in one-shot}},
  volume       = {202},
  year         = {2023},
}

@article{12702,
  abstract     = {Hydrocarbon mixtures are extremely abundant in the Universe, and diamond formation from them can play a crucial role in shaping the interior structure and evolution of planets. With first-principles accuracy, we first estimate the melting line of diamond, and then reveal the nature of chemical bonding in hydrocarbons at extreme conditions. We finally establish the pressure-temperature phase boundary where it is thermodynamically possible for diamond to form from hydrocarbon mixtures with different atomic fractions of carbon. Notably, here we show a depletion zone at pressures above 200 GPa and temperatures below 3000 K-3500 K where diamond formation is thermodynamically favorable regardless of the carbon atomic fraction, due to a phase separation mechanism. The cooler condition of the interior of Neptune compared to Uranus means that the former is much more likely to contain the depletion zone. Our findings can help explain the dichotomy of the two ice giants manifested by the low luminosity of Uranus, and lead to a better understanding of (exo-)planetary formation and evolution.},
  author       = {Cheng, Bingqing and Hamel, Sebastien and Bethkenhagen, Mandy},
  issn         = {2041-1723},
  journal      = {Nature Communications},
  publisher    = {Springer Nature},
  title        = {{Thermodynamics of diamond formation from hydrocarbon mixtures in planets}},
  doi          = {10.1038/s41467-023-36841-1},
  volume       = {14},
  year         = {2023},
}

@article{12912,
  abstract     = {The chemical potential of adsorbed or confined fluids provides insight into their unique thermodynamic properties and determines adsorption isotherms. However, it is often difficult to compute this quantity from atomistic simulations using existing statistical mechanical methods. We introduce a computational framework that utilizes static structure factors, thermodynamic integration, and free energy perturbation for calculating the absolute chemical potential of fluids. For demonstration, we apply the method to compute the adsorption isotherms of carbon dioxide in a metal-organic framework and water in carbon nanotubes.},
  author       = {Schmid, Rochus and Cheng, Bingqing},
  issn         = {1089-7690},
  journal      = {The Journal of Chemical Physics},
  number       = {16},
  publisher    = {AIP Publishing},
  title        = {{Computing chemical potentials of adsorbed or confined fluids}},
  doi          = {10.1063/5.0146711},
  volume       = {158},
  year         = {2023},
}

@article{12879,
  abstract     = {Machine learning (ML) has been widely applied to chemical property prediction, most prominently for the energies and forces in molecules and materials. The strong interest in predicting energies in particular has led to a ‘local energy’-based paradigm for modern atomistic ML models, which ensures size-extensivity and a linear scaling of computational cost with system size. However, many electronic properties (such as excitation energies or ionization energies) do not necessarily scale linearly with system size and may even be spatially localized. Using size-extensive models in these cases can lead to large errors. In this work, we explore different strategies for learning intensive and localized properties, using HOMO energies in organic molecules as a representative test case. In particular, we analyze the pooling functions that atomistic neural networks use to predict molecular properties, and suggest an orbital weighted average (OWA) approach that enables the accurate prediction of orbital energies and locations.},
  author       = {Chen, Ke and Kunkel, Christian and Cheng, Bingqing and Reuter, Karsten and Margraf, Johannes T.},
  issn         = {2041-6539},
  journal      = {Chemical Science},
  publisher    = {Royal Society of Chemistry},
  title        = {{Physics-inspired machine learning of localized intensive properties}},
  doi          = {10.1039/d3sc00841j},
  year         = {2023},
}

@article{14425,
  abstract     = {Water adsorption and dissociation processes on pristine low-index TiO2 interfaces are important but poorly understood outside the well-studied anatase (101) and rutile (110). To understand these, we construct three sets of machine learning potentials that are simultaneously applicable to various TiO2 surfaces, based on three density-functional-theory approximations. Here we show the water dissociation free energies on seven pristine TiO2 surfaces, and predict that anatase (100), anatase (110), rutile (001), and rutile (011) favor water dissociation, anatase (101) and rutile (100) have mostly molecular adsorption, while the simulations of rutile (110) sensitively depend on the slab thickness and molecular adsorption is preferred with thick slabs. Moreover, using an automated algorithm, we reveal that these surfaces follow different types of atomistic mechanisms for proton transfer and water dissociation: one-step, two-step, or both. These mechanisms can be rationalized based on the arrangements of water molecules on the different surfaces. Our finding thus demonstrates that the different pristine TiO2 surfaces react with water in distinct ways, and cannot be represented using just the low-energy anatase (101) and rutile (110) surfaces.},
  author       = {Zeng, Zezhu and Wodaczek, Felix and Liu, Keyang and Stein, Frederick and Hutter, Jürg and Chen, Ji and Cheng, Bingqing},
  issn         = {2041-1723},
  journal      = {Nature Communications},
  publisher    = {Springer Nature},
  title        = {{Mechanistic insight on water dissociation on pristine low-index TiO2 surfaces from machine learning molecular dynamics simulations}},
  doi          = {10.1038/s41467-023-41865-8},
  volume       = {14},
  year         = {2023},
}

@article{14603,
  abstract     = {Computing the solubility of crystals in a solvent using atomistic simulations is notoriously challenging due to the complexities and convergence issues associated with free-energy methods, as well as the slow equilibration in direct-coexistence simulations. This paper introduces a molecular-dynamics workflow that simplifies and robustly computes the solubility of molecular or ionic crystals. This method is considerably more straightforward than the state-of-the-art, as we have streamlined and optimised each step of the process. Specifically, we calculate the chemical potential of the crystal using the gas-phase molecule as a reference state, and employ the S0 method to determine the concentration dependence of the chemical potential of the solute. We use this workflow to predict the solubilities of sodium chloride in water, urea polymorphs in water, and paracetamol polymorphs in both water and ethanol. Our findings indicate that the predicted solubility is sensitive to the chosen potential energy surface. Furthermore, we note that the harmonic approximation often fails for both molecular crystals and gas molecules at or above room temperature, and that the assumption of an ideal solution becomes less valid for highly soluble substances.},
  author       = {Reinhardt, Aleks and Chew, Pin Yu and Cheng, Bingqing},
  issn         = {1089-7690},
  journal      = {Journal of Chemical Physics},
  number       = {18},
  publisher    = {AIP Publishing},
  title        = {{A streamlined molecular-dynamics workflow for computing solubilities of molecular and ionic crystals}},
  doi          = {10.1063/5.0173341},
  volume       = {159},
  year         = {2023},
}

@article{13231,
  abstract     = {We study ab initio approaches for calculating x-ray Thomson scattering spectra from density functional theory molecular dynamics simulations based on a modified Chihara formula that expresses the inelastic contribution in terms of the dielectric function. We study the electronic dynamic structure factor computed from the Mermin dielectric function using an ab initio electron-ion collision frequency in comparison to computations using a linear-response time-dependent density functional theory (LR-TDDFT) framework for hydrogen and beryllium and investigate the dispersion of free-free and bound-free contributions to the scattering signal. A separate treatment of these contributions, where only the free-free part follows the Mermin dispersion, shows good agreement with LR-TDDFT results for ambient-density beryllium, but breaks down for highly compressed matter where the bound states become pressure ionized. LR-TDDFT is used to reanalyze x-ray Thomson scattering experiments on beryllium demonstrating strong deviations from the plasma conditions inferred with traditional analytic models at small scattering angles.},
  author       = {Schörner, Maximilian and Bethkenhagen, Mandy and Döppner, Tilo and Kraus, Dominik and Fletcher, Luke B. and Glenzer, Siegfried H. and Redmer, Ronald},
  issn         = {2470-0053},
  journal      = {Physical Review E},
  number       = {6},
  publisher    = {American Physical Society},
  title        = {{X-ray Thomson scattering spectra from density functional theory molecular dynamics simulations based on a modified Chihara formula}},
  doi          = {10.1103/PhysRevE.107.065207},
  volume       = {107},
  year         = {2023},
}

@article{14605,
  abstract     = {The phonon transport mechanisms and ultralow lattice thermal conductivities (κL) in silver halide AgX (X=Cl,Br,I) compounds are not yet well understood. Herein, we study the lattice dynamics and thermal property of AgX under the framework of perturbation theory and the two-channel Wigner thermal transport model based on accurate machine learning potentials. We find that an accurate extraction of the third-order atomic force constants from largely displaced configurations is significant for the calculation of the κL of AgX, and the coherence thermal transport is also non-negligible. In AgI, however, the calculated κL still considerably overestimates the experimental values even including four-phonon scatterings. Molecular dynamics (MD) simulations using machine learning potential suggest an important role of the higher-than-fourth-order lattice anharmonicity in the low-frequency phonon linewidths of AgI at room temperature, which can be related to the simultaneous restrictions of the three- and four-phonon phase spaces. The κL of AgI calculated using MD phonon lifetimes including full-order lattice anharmonicity shows a better agreement with experiments.},
  author       = {Ouyang, Niuchang and Zeng, Zezhu and Wang, Chen and Wang, Qi and Chen, Yue},
  issn         = {2469-9969},
  journal      = {Physical Review B},
  number       = {17},
  publisher    = {American Physical Society},
  title        = {{Role of high-order lattice anharmonicity in the phonon thermal transport of silver halide AgX (X=Cl,Br, I)}},
  doi          = {10.1103/PhysRevB.108.174302},
  volume       = {108},
  year         = {2023},
}

@misc{14619,
  abstract     = {Data underlying the publication "A streamlined molecular-dynamics workflow for computing solubilities of molecular and ionic crystals" (DOI https://doi.org/10.1063/5.0173341).},
  author       = {Cheng, Bingqing},
  publisher    = {Zenodo},
  title        = {{BingqingCheng/solubility: V1.0}},
  doi          = {10.5281/ZENODO.8398094},
  year         = {2023},
}

@article{13216,
  abstract     = {Physical catalysts often have multiple sites where reactions can take place. One prominent example is single-atom alloys, where the reactive dopant atoms can preferentially locate in the bulk or at different sites on the surface of the nanoparticle. However, ab initio modeling of catalysts usually only considers one site of the catalyst, neglecting the effects of multiple sites. Here, nanoparticles of copper doped with single-atom rhodium or palladium are modeled for the dehydrogenation of propane. Single-atom alloy nanoparticles are simulated at 400–600 K, using machine learning potentials trained on density functional theory calculations, and then the occupation of different single-atom active sites is identified using a similarity kernel. Further, the turnover frequency for all possible sites is calculated for propane dehydrogenation to propene through microkinetic modeling using density functional theory calculations. The total turnover frequencies of the whole nanoparticle are then described from both the population and the individual turnover frequency of each site. Under operating conditions, rhodium as a dopant is found to almost exclusively occupy (111) surface sites while palladium as a dopant occupies a greater variety of facets. Undercoordinated dopant surface sites are found to tend to be more reactive for propane dehydrogenation compared to the (111) surface. It is found that considering the dynamics of the single-atom alloy nanoparticle has a profound effect on the calculated catalytic activity of single-atom alloys by several orders of magnitude.},
  author       = {Bunting, Rhys and Wodaczek, Felix and Torabi, Tina and Cheng, Bingqing},
  issn         = {1520-5126},
  journal      = {Journal of the American Chemical Society},
  keywords     = {Colloid and Surface Chemistry, Biochemistry, General Chemistry, Catalysis},
  number       = {27},
  pages        = {14894--14902},
  publisher    = {American Chemical Society},
  title        = {{Reactivity of single-atom alloy nanoparticles: Modeling the dehydrogenation of propane}},
  doi          = {10.1021/jacs.3c04030},
  volume       = {145},
  year         = {2023},
}

@article{13118,
  abstract     = {Under high pressures and temperatures, molecular systems with substantial polarization charges, such as ammonia and water, are predicted to form superionic phases and dense fluid states with dissociating molecules and high electrical conductivity. This behaviour potentially plays a role in explaining the origin of the multipolar magnetic fields of Uranus and Neptune, whose mantles are thought to result from a mixture of H2O, NH3 and CH4 ices. Determining the stability domain, melting curve and electrical conductivity of these superionic phases is therefore crucial for modelling planetary interiors and dynamos. Here we report the melting curve of superionic ammonia up to 300 GPa from laser-driven shock compression of pre-compressed samples and atomistic calculations. We show that ammonia melts at lower temperatures than water above 100 GPa and that fluid ammonia’s electrical conductivity exceeds that of water at conditions predicted by hot, super-adiabatic models for Uranus and Neptune, and enhances the conductivity in their fluid water-rich dynamo layers.},
  author       = {Hernandez, J.-A. and Bethkenhagen, Mandy and Ninet, S. and French, M. and Benuzzi-Mounaix, A. and Datchi, F. and Guarguaglini, M. and Lefevre, F. and Occelli, F. and Redmer, R. and Vinci, T. and Ravasio, A.},
  issn         = {1745-2481},
  journal      = {Nature Physics},
  pages        = {1280--1285},
  publisher    = {Springer Nature},
  title        = {{Melting curve of superionic ammonia at planetary interior conditions}},
  doi          = {10.1038/s41567-023-02074-8},
  volume       = {19},
  year         = {2023},
}

@article{13039,
  abstract     = {We calculate reflectivities of dynamically compressed water, water-ethanol mixtures, and ammonia at infrared and optical wavelengths with density functional theory and molecular dynamics simulations. The influence of the exchange-correlation functional on the results is examined in detail. Our findings indicate that the consistent use of the HSE hybrid functional reproduces experimental results much better than the commonly used PBE functional. The HSE functional offers not only a more accurate description of the electronic band gap but also shifts the onset of molecular dissociation in the molecular dynamics simulations to significantly higher pressures. We also highlight the importance of using accurate reference standards in reflectivity experiments and reanalyze infrared and optical reflectivity data from recent experiments. Thus, our combined theoretical and experimental work explains and resolves lingering discrepancies between calculations and measurements for the investigated molecular substances under shock compression.},
  author       = {French, Martin and Bethkenhagen, Mandy and Ravasio, Alessandra and Hernandez, Jean Alexis},
  issn         = {2469-9969},
  journal      = {Physical Review B},
  number       = {13},
  publisher    = {American Physical Society},
  title        = {{Ab initio calculation of the reflectivity of molecular fluids under shock compression}},
  doi          = {10.1103/PhysRevB.107.134109},
  volume       = {107},
  year         = {2023},
}

@article{13988,
  abstract     = {Most permissionless blockchains inherently suffer from throughput limitations. Layer-2 systems, such as side-chains or Rollups, have been proposed as a possible strategy to overcome this limitation. Layer-2 systems interact with the main-chain in two ways. First, users can move funds from/to the main-chain to/from the layer-2. Second, layer-2 systems periodically synchronize with the main-chain to keep some form of log of their activity on the main-chain - this log is key for security. Due to this interaction with the main-chain, which is necessary and recurrent, layer-2 systems impose some load on the main-chain. The impact of such load on the main-chain has been, so far, poorly understood. In addition to that, layer-2 approaches typically sacrifice decentralization and security in favor of higher throughput. This paper presents an experimental study that analyzes the current state of Ethereum layer-2 projects. Our goal is to assess the load they impose on Ethereum and to understand their scalability potential in the long-run. Our analysis shows that the impact of any given layer-2 on the main-chain is the result of both technical aspects (how state is logged on the main-chain) and user behavior (how often users decide to transfer funds between the layer-2 and the main-chain). Based on our observations, we infer that without efficient mechanisms that allow users to transfer funds in a secure and fast manner directly from one layer-2 project to another, current layer-2 systems will not be able to scale Ethereum effectively, regardless of their technical solutions. Furthermore, from our results, we conclude that the layer-2 systems that offer similar security guarantees as Ethereum have limited scalability potential, while approaches that offer better performance, sacrifice security and lead to an increase in centralization which runs against the end-goals of permissionless blockchains.},
  author       = {Neiheiser, Ray and Inacio, Gustavo and Rech, Luciana and Montez, Carlos and Matos, Miguel and Rodrigues, Luis},
  issn         = {2169-3536},
  journal      = {IEEE Access},
  keywords     = {General Engineering, General Materials Science, General Computer Science, Electrical and Electronic Engineering},
  pages        = {8651--8662},
  publisher    = {IEEE},
  title        = {{Practical limitations of Ethereum’s layer-2}},
  doi          = {10.1109/access.2023.3237897},
  volume       = {11},
  year         = {2023},
}

@phdthesis{13074,
  abstract     = {Deep learning has become an integral part of a large number of important applications, and many of the recent breakthroughs have been enabled by the ability to train very large models, capable to capture complex patterns and relationships from the data. At the same time, the massive sizes of modern deep learning models have made their deployment to smaller devices more challenging; this is particularly important, as in many applications the users rely on accurate deep learning predictions, but they only have access to devices with limited memory and compute power. One solution to this problem is to prune neural networks, by setting as many of their parameters as possible to zero, to obtain accurate sparse models with lower memory footprint. Despite the great research progress in obtaining sparse models that preserve accuracy, while satisfying memory and computational constraints, there are still many challenges associated with efficiently training sparse models, as well as understanding their generalization properties.

The focus of this thesis is to investigate how the training process of sparse models can be made more efficient, and to understand the differences between sparse and dense models in terms of how well they can generalize to changes in the data distribution. We first study a method for co-training sparse and dense models, at a lower cost compared to regular training. With our method we can obtain very accurate sparse networks, and dense models that can recover the baseline accuracy. Furthermore, we are able to more easily analyze the differences, at prediction level, between the sparse-dense model pairs. Next, we investigate the generalization properties of sparse neural networks in more detail, by studying how well different sparse models trained on a larger task can adapt to smaller, more specialized tasks, in a transfer learning scenario. Our analysis across multiple pruning methods and sparsity levels reveals that sparse models provide features that can transfer similarly to or better than the dense baseline. However, the choice of the pruning method plays an important role, and can influence the results when the features are fixed (linear finetuning), or when they are allowed to adapt to the new task (full finetuning). Using sparse models with fixed masks for finetuning on new tasks has an important practical advantage, as it enables training neural networks on smaller devices. However, one drawback of current pruning methods is that the entire training cycle has to be repeated to obtain the initial sparse model, for every sparsity target; in consequence, the entire training process is costly and also multiple models need to be stored. In the last part of the thesis we propose a method that can train accurate dense models that are compressible in a single step, to multiple sparsity levels, without additional finetuning. Our method results in sparse models that can be competitive with existing pruning methods, and which can also successfully generalize to new tasks.},
  author       = {Peste, Elena-Alexandra},
  issn         = {2663-337X},
  pages        = {147},
  publisher    = {Institute of Science and Technology Austria},
  title        = {{Efficiency and generalization of sparse neural networks}},
  doi          = {10.15479/at:ista:13074},
  year         = {2023},
}

@inproceedings{12548,
  abstract     = {The limited exchange between human communities is a key factor in preventing the spread of COVID-19. This paper introduces a digital framework that combines an integration of real mobility data at the country scale with a series of modeling techniques and visual capabilities that highlight mobility patterns before and during the pandemic. The findings not only significantly exhibit mobility trends and different degrees of similarities at regional and local levels but also provide potential insight into the emergence of a pandemic on human behavior patterns and their likely socio-economic impacts.},
  author       = {Forghani, Mohammad and Claramunt, Christophe and Karimipour, Farid and Heiler, Georg},
  booktitle    = {2022 IEEE International Conference on Data Mining Workshops},
  issn         = {2375-9259},
  location     = {Orlando, FL, United States},
  publisher    = {IEEE},
  title        = {{Visual analytics of mobility network changes observed using mobile phone data during COVID-19 pandemic}},
  doi          = {10.1109/icdmw58026.2022.00093},
  year         = {2023},
}

@inproceedings{13321,
  abstract     = {We consider the problem of reconstructing the signal and the hidden variables from observations coming from a multi-layer network with rotationally invariant weight matrices. The multi-layer structure models inference from deep generative priors, and the rotational invariance imposed on the weights generalizes the i.i.d. Gaussian assumption by allowing for a complex correlation structure, which is typical in applications. In this work, we present a new class of approximate message passing (AMP) algorithms and give a state evolution recursion which precisely characterizes their performance in the large system limit. In contrast with the existing multi-layer VAMP (ML-VAMP) approach, our proposed AMP – dubbed multilayer rotationally invariant generalized AMP (ML-RI-GAMP) – provides a natural generalization beyond Gaussian designs, in the sense that it recovers the existing Gaussian AMP as a special case. Furthermore, ML-RI-GAMP exhibits a significantly lower complexity than ML-VAMP, as the computationally intensive singular value decomposition is replaced by an estimation of the moments of the design matrices. Finally, our numerical results show that this complexity gain comes at little to no cost in the performance of the algorithm.},
  author       = {Xu, Yizhou and Hou, Tian Qi and Liang, Shan Suo and Mondelli, Marco},
  booktitle    = {2023 IEEE Information Theory Workshop},
  isbn         = {9798350301496},
  issn         = {2475-4218},
  location     = {Saint-Malo, France},
  pages        = {294--298},
  publisher    = {IEEE},
  title        = {{Approximate message passing for multi-layer estimation in rotationally invariant models}},
  doi          = {10.1109/ITW55543.2023.10160238},
  year         = {2023},
}

@article{12704,
  abstract     = {Adversarial training (i.e., training on adversarially perturbed input data) is a well-studied method for making neural networks robust to potential adversarial attacks during inference. However, the improved robustness does not come for free but rather is accompanied by a decrease in overall model accuracy and performance. Recent work has shown that, in practical robot learning applications, the effects of adversarial training do not pose a fair trade-off but inflict a net loss when measured in holistic robot performance. This work revisits the robustness-accuracy trade-off in robot learning by systematically analyzing if recent advances in robust training methods and theory in conjunction with adversarial robot learning, are capable of making adversarial training suitable for real-world robot applications. We evaluate three different robot learning tasks ranging from autonomous driving in a high-fidelity environment amenable to sim-to-real deployment to mobile robot navigation and gesture recognition. Our results demonstrate that, while these techniques make incremental improvements on the trade-off on a relative scale, the negative impact on the nominal accuracy caused by adversarial training still outweighs the improved robustness by an order of magnitude. We conclude that although progress is happening, further advances in robust learning methods are necessary before they can benefit robot learning tasks in practice.},
  author       = {Lechner, Mathias and Amini, Alexander and Rus, Daniela and Henzinger, Thomas A},
  issn         = {2377-3766},
  journal      = {IEEE Robotics and Automation Letters},
  number       = {3},
  pages        = {1595--1602},
  publisher    = {IEEE},
  title        = {{Revisiting the adversarial robustness-accuracy tradeoff in robot learning}},
  doi          = {10.1109/LRA.2023.3240930},
  volume       = {8},
  year         = {2023},
}

@inproceedings{13967,
  abstract     = {A classic solution technique for Markov decision processes (MDP) and stochastic games (SG) is value iteration (VI). Due to its good practical performance, this approximative approach is typically preferred over exact techniques, even though no practical bounds on the imprecision of the result could be given until recently. As a consequence, even the most used model checkers could return arbitrarily wrong results. Over the past decade, different works derived stopping criteria, indicating when the precision reaches the desired level, for various settings, in particular MDP with reachability, total reward, and mean payoff, and SG with reachability.In this paper, we provide the first stopping criteria for VI on SG with total reward and mean payoff, yielding the first anytime algorithms in these settings. To this end, we provide the solution in two flavours: First through a reduction to the MDP case and second directly on SG. The former is simpler and automatically utilizes any advances on MDP. The latter allows for more local computations, heading towards better practical efficiency.Our solution unifies the previously mentioned approaches for MDP and SG and their underlying ideas. To achieve this, we isolate objective-specific subroutines as well as identify objective-independent concepts. These structural concepts, while surprisingly simple, form the very essence of the unified solution.},
  author       = {Kretinsky, Jan and Meggendorfer, Tobias and Weininger, Maximilian},
  booktitle    = {38th Annual ACM/IEEE Symposium on Logic in Computer Science},
  isbn         = {9798350335873},
  issn         = {1043-6871},
  location     = {Boston, MA, United States},
  publisher    = {IEEE},
  title        = {{Stopping criteria for value iteration on stochastic games with quantitative objectives}},
  doi          = {10.1109/LICS56636.2023.10175771},
  volume       = {2023},
  year         = {2023},
}

@article{14751,
  abstract     = {We consider zero-error communication over a two-transmitter deterministic adversarial multiple access channel (MAC) governed by an adversary who has access to the transmissions of both senders (hence called omniscient ) and aims to maliciously corrupt the communication. None of the encoders, jammer and decoder is allowed to randomize using private or public randomness. This enforces a combinatorial nature of the problem. Our model covers a large family of channels studied in the literature, including all deterministic discrete memoryless noisy or noiseless MACs. In this work, given an arbitrary two-transmitter deterministic omniscient adversarial MAC, we characterize when the capacity region: 1) has nonempty interior (in particular, is two-dimensional); 2) consists of two line segments (in particular, has empty interior); 3) consists of one line segment (in particular, is one-dimensional); 4) or only contains (0,0) (in particular, is zero-dimensional). This extends a recent result by Wang et al. (201 9) from the point-to-point setting to the multiple access setting. Indeed, our converse arguments build upon their generalized Plotkin bound and involve delicate case analysis. One of the technical challenges is to take care of both “joint confusability” and “marginal confusability”. In particular, the treatment of marginal confusability does not follow from the point-to-point results by Wang et al. Our achievability results follow from random coding with expurgation.},
  author       = {Zhang, Yihan},
  issn         = {1557-9654},
  journal      = {IEEE Transactions on Information Theory},
  keywords     = {Computer Science Applications, Information Systems},
  number       = {7},
  pages        = {4093--4127},
  publisher    = {IEEE},
  title        = {{Zero-error communication over adversarial MACs}},
  doi          = {10.1109/tit.2023.3257239},
  volume       = {69},
  year         = {2023},
}

