@unpublished{50172,
  abstract     = {{Viscous hydrodynamics serves as a successful mesoscopic description of the
Quark-Gluon Plasma produced in relativistic heavy-ion collisions. In order to
investigate, how such an effective description emerges from the underlying
microscopic dynamics we calculate the hydrodynamic and non-hydrodynamic modes
of linear response in the sound channel from a first-principle calculation in
kinetic theory. We do this with a new approach wherein we discretize the
collision kernel to directly calculate eigenvalues and eigenmodes of the
evolution operator. This allows us to study the Green's functions at any point
in the complex frequency space. Our study focuses on scalar theory with quartic
interaction and we find that the analytic structure of Green's functions in the
complex plane is far more complicated than just poles or cuts which is a first
step towards an equivalent study in QCD kinetic theory.}},
  author       = {{Ochsenfeld, Stephan and Schlichting, Sören}},
  booktitle    = {{arXiv:2308.04491}},
  title        = {{{Hydrodynamic and Non-hydrodynamic Excitations in Kinetic Theory -- A  Numerical Analysis in Scalar Field Theory}}},
  year         = {{2023}},
}

@unpublished{50221,
  abstract     = {{Memory Gym presents a suite of 2D partially observable environments, namely
Mortar Mayhem, Mystery Path, and Searing Spotlights, designed to benchmark
memory capabilities in decision-making agents. These environments, originally
with finite tasks, are expanded into innovative, endless formats, mirroring the
escalating challenges of cumulative memory games such as ``I packed my bag''.
This progression in task design shifts the focus from merely assessing sample
efficiency to also probing the levels of memory effectiveness in dynamic,
prolonged scenarios. To address the gap in available memory-based Deep
Reinforcement Learning baselines, we introduce an implementation that
integrates Transformer-XL (TrXL) with Proximal Policy Optimization. This
approach utilizes TrXL as a form of episodic memory, employing a sliding window
technique. Our comparative study between the Gated Recurrent Unit (GRU) and
TrXL reveals varied performances across different settings. TrXL, on the finite
environments, demonstrates superior sample efficiency in Mystery Path and
outperforms in Mortar Mayhem. However, GRU is more efficient on Searing
Spotlights. Most notably, in all endless tasks, GRU makes a remarkable
resurgence, consistently outperforming TrXL by significant margins. Website and
Source Code: https://github.com/MarcoMeter/endless-memory-gym/}},
  author       = {{Pleines, Marco and Pallasch, Matthias and Zimmer, Frank and Preuss, Mike}},
  booktitle    = {{arXiv:2309.17207}},
  title        = {{{Memory Gym: Towards Endless Tasks to Benchmark Memory Capabilities of  Agents}}},
  year         = {{2023}},
}

@unpublished{43439,
  abstract     = {{This preprint makes the claim of having computed the $9^{th}$ Dedekind
Number. This was done by building an efficient FPGA Accelerator for the core
operation of the process, and parallelizing it on the Noctua 2 Supercluster at
Paderborn University. The resulting value is
286386577668298411128469151667598498812366. This value can be verified in two
steps. We have made the data file containing the 490M results available, each
of which can be verified separately on CPU, and the whole file sums to our
proposed value.}},
  author       = {{Van Hirtum, Lennart and De Causmaecker, Patrick and Goemaere, Jens and Kenter, Tobias and Riebler, Heinrich and Lass, Michael and Plessl, Christian}},
  booktitle    = {{arXiv:2304.03039}},
  title        = {{{A computation of D(9) using FPGA Supercomputing}}},
  year         = {{2023}},
}

@inproceedings{46188,
  author       = {{Faj, Jennifer and Kenter, Tobias and Faghih-Naini, Sara and Plessl, Christian and Aizinger, Vadym}},
  booktitle    = {{Proceedings of the Platform for Advanced Scientific Computing Conference (PASC)}},
  publisher    = {{ACM}},
  title        = {{{Scalable Multi-FPGA Design of a Discontinuous Galerkin Shallow-Water Model on Unstructured Meshes}}},
  doi          = {{10.1145/3592979.3593407}},
  year         = {{2023}},
}

@inproceedings{46189,
  author       = {{Prouveur, Charles and Haefele, Matthieu and Kenter, Tobias and Voss, Nils}},
  booktitle    = {{Proceedings of the Platform for Advanced Scientific Computing Conference (PASC)}},
  publisher    = {{ACM}},
  title        = {{{FPGA Acceleration for HPC Supercapacitor Simulations}}},
  doi          = {{10.1145/3592979.3593419}},
  year         = {{2023}},
}

@inbook{45893,
  author       = {{Hansmeier, Tim and Kenter, Tobias and Meyer, Marius and Riebler, Heinrich and Platzner, Marco and Plessl, Christian}},
  booktitle    = {{On-The-Fly Computing -- Individualized IT-services in dynamic markets}},
  editor       = {{Haake, Claus-Jochen and Meyer auf der Heide, Friedhelm and Platzner, Marco and Wachsmuth, Henning and Wehrheim, Heike}},
  pages        = {{165--182}},
  publisher    = {{Heinz Nixdorf Institut, Universität Paderborn}},
  title        = {{{Compute Centers I: Heterogeneous Execution Environments}}},
  doi          = {{10.5281/zenodo.8068642}},
  volume       = {{412}},
  year         = {{2023}},
}

@article{54854,
  abstract     = {{<jats:p>Batteries based on heavier alkali ions are considered promising candidates to substitute for current Li-based technologies. In this theoretical study, we characterize the structural properties of a novel material, i.e., F-doped RbTiOPO4 (RbTiPO4F, RTP:F), and discuss aspects of its electrochemical performance in Rb-ion batteries (RIBs) using density functional theory (DFT). According to our calculations, RTP:F is expected to retain the so-called KTiOPO4 (KTP)-type structure, with lattice parameters of 13.236 Å, 6.616 Å, and 10.945 Å. Due to the doping with F, the crystal features eight extra electrons per unit cell, whereby each of these electrons is trapped by one of the surrounding Ti atoms in the cell. Notably, the ground state of the system corresponds to a ferromagnetic spin configuration (i.e., S=4). The deintercalation of Rb leads to the oxidation of the Ti atoms in the cell (i.e., from Ti3+ to Ti4+) and to reduced magnetic moments. The material promises interesting electrochemical properties for the cathode: rather high average voltages above 2.8 V and modest volume shrinkages below 13% even in the fully deintercalated case are predicted.</jats:p>}},
  author       = {{Bocchini, Adriana and Xie, Yingjie and Schmidt, Wolf Gero and Gerstmann, Uwe}},
  issn         = {{2073-4352}},
  journal      = {{Crystals}},
  number       = {{1}},
  publisher    = {{MDPI AG}},
  title        = {{{Structural and Electrochemical Properties of F-Doped RbTiOPO4 (RTP:F) Predicted from First Principles}}},
  doi          = {{10.3390/cryst14010005}},
  volume       = {{14}},
  year         = {{2023}},
}

@article{54853,
  abstract     = {{<jats:p>The nitrogen-vacancy (NV) centers (NCVSi)− in 4H silicon carbide (SiC) constitute an ensemble of spin S = 1 solid state qubits interacting with the surrounding 14N and 29Si nuclei. As quantum applications based on a polarization transfer from the electron spin to the nuclei require the knowledge of the electron–nuclear interaction parameters, we have used high-frequency (94 GHz) electron–nuclear double resonance spectroscopy combined with first-principles density functional theory to investigate the hyperfine and nuclear quadrupole interactions of the basal and axial NV centers. We observed that the four inequivalent NV configurations (hk, kh, hh, and kk) exhibit different electron–nuclear interaction parameters, suggesting that each NV center may act as a separate optically addressable qubit. Finally, we rationalized the observed differences in terms of distinctions in the local atomic structures of the NV configurations. Thus, our results provide the basic knowledge for an extension of quantum protocols involving the 14N nuclear spin.</jats:p>}},
  author       = {{Murzakhanov, F. F. and Sadovnikova, M. A. and Mamin, G. V. and Nagalyuk, S. S. and von Bardeleben, H. J. and Schmidt, Wolf Gero and Biktagirov, Timur and Gerstmann, Uwe and Soltamov, V. A.}},
  issn         = {{0021-8979}},
  journal      = {{Journal of Applied Physics}},
  number       = {{12}},
  publisher    = {{AIP Publishing}},
  title        = {{{14N Hyperfine and nuclear interactions of axial and basal NV centers in 4H-SiC: A high frequency (94 GHz) ENDOR study}}},
  doi          = {{10.1063/5.0170099}},
  volume       = {{134}},
  year         = {{2023}},
}

@article{54851,
  abstract     = {{<jats:p>Composites of different graphene oxide types, TiO<jats:sub>2</jats:sub> materials, and especially synthetic routes influence the photocatalytic activity of the resulting material.</jats:p>}},
  author       = {{Rosenthal, Marta and Biktagirov, Timur and Schmidt, Wolf Gero and Wilhelm, René}},
  issn         = {{2044-4753}},
  journal      = {{Catalysis Science &amp; Technology}},
  number       = {{15}},
  pages        = {{4367--4377}},
  publisher    = {{Royal Society of Chemistry (RSC)}},
  title        = {{{Synthesis of new graphene oxide/TiO<sub>2</sub> and TiO<sub>2</sub>/SiO<sub>2</sub> nanocomposites and their evaluation as photocatalysts}}},
  doi          = {{10.1039/d3cy00461a}},
  volume       = {{13}},
  year         = {{2023}},
}

@article{54850,
  author       = {{Meier, Lukas and Schmidt, Wolf Gero}},
  issn         = {{1932-7447}},
  journal      = {{The Journal of Physical Chemistry C}},
  number       = {{4}},
  pages        = {{1973--1980}},
  publisher    = {{American Chemical Society (ACS)}},
  title        = {{{Adsorption of Cyclic (Alkyl) (Amino) Carbenes on Monohydride Si(001) Surfaces: Interface Bonding and Electronic Properties}}},
  doi          = {{10.1021/acs.jpcc.2c07316}},
  volume       = {{127}},
  year         = {{2023}},
}

@article{46120,
  abstract     = {{The rise of exascale supercomputers has fueled competition among GPU vendors, driving lattice QCD developers to write code that supports multiple APIs. Moreover, new developments in algorithms and physics research require frequent updates to existing software. These challenges have to be balanced against constantly changing personnel. At the same time, there is a wide range of applications for HISQ fermions in QCD studies. This situation encourages the development of software featuring a HISQ action that is flexible, high-performing, open source, easy to use, and easy to adapt. In this technical paper, we explain the design strategy, provide implementation details, list available algorithms and modules, and show key performance indicators for SIMULATeQCD, a simple multi-GPU lattice code for large-scale QCD calculations, mainly developed and used by the HotQCD collaboration. The code is publicly available on GitHub.}},
  author       = {{Mazur, Lukas and Bollweg, Dennis and Clarke, David A. and Altenkort, Luis and Kaczmarek, Olaf and Larsen, Rasmus and Shu, Hai-Tao and Goswami, Jishnu and Scior, Philipp and Sandmeyer, Hauke and Neumann, Marius and Dick, Henrik and Ali, Sajid and Kim, Jangho and Schmidt, Christian and Petreczky, Peter and Mukherjee, Swagato}},
  journal      = {{Computer Physics Communications}},
  title        = {{{SIMULATeQCD: A simple multi-GPU lattice code for QCD calculations}}},
  doi          = {{10.48550/ARXIV.2306.01098}},
  year         = {{2023}},
}

@article{46119,
  author       = {{Altenkort, Luis and Eller, Alexander M. and Francis, Anthony and Kaczmarek, Olaf and Mazur, Lukas and Moore, Guy D. and Shu, Hai-Tao}},
  issn         = {{2470-0010}},
  journal      = {{Physical Review D}},
  number       = {{1}},
  publisher    = {{American Physical Society (APS)}},
  title        = {{{Viscosity of pure-glue QCD from the lattice}}},
  doi          = {{10.1103/physrevd.108.014503}},
  volume       = {{108}},
  year         = {{2023}},
}

@article{38041,
  abstract     = {{<jats:p>While FPGA accelerator boards and their respective high-level design tools are maturing, there is still a lack of multi-FPGA applications, libraries, and not least, benchmarks and reference implementations towards sustained HPC usage of these devices. As in the early days of GPUs in HPC, for workloads that can reasonably be decoupled into loosely coupled working sets, multi-accelerator support can be achieved by using standard communication interfaces like MPI on the host side. However, for performance and productivity, some applications can profit from a tighter coupling of the accelerators. FPGAs offer unique opportunities here when extending the dataflow characteristics to their communication interfaces.</jats:p>
          <jats:p>In this work, we extend the HPCC FPGA benchmark suite by multi-FPGA support and three missing benchmarks that particularly characterize or stress inter-device communication: b_eff, PTRANS, and LINPACK. With all benchmarks implemented for current boards with Intel and Xilinx FPGAs, we established a baseline for multi-FPGA performance. Additionally, for the communication-centric benchmarks, we explored the potential of direct FPGA-to-FPGA communication with a circuit-switched inter-FPGA network that is currently only available for one of the boards. The evaluation with parallel execution on up to 26 FPGA boards makes use of one of the largest academic FPGA installations.</jats:p>}},
  author       = {{Meyer, Marius and Kenter, Tobias and Plessl, Christian}},
  issn         = {{1936-7406}},
  journal      = {{ACM Transactions on Reconfigurable Technology and Systems}},
  keywords     = {{General Computer Science}},
  publisher    = {{Association for Computing Machinery (ACM)}},
  title        = {{{Multi-FPGA Designs and Scaling of HPC Challenge Benchmarks via MPI and Circuit-Switched Inter-FPGA Networks}}},
  doi          = {{10.1145/3576200}},
  year         = {{2023}},
}

@inproceedings{43228,
  abstract     = {{The computation of electron repulsion integrals (ERIs) over Gaussian-type orbitals (GTOs) is a challenging problem in quantum-mechanics-based atomistic simulations. In practical simulations, several trillions of ERIs may have to be
computed for every time step.
In this work, we investigate FPGAs as accelerators for the ERI computation. We use template parameters, here within the Intel oneAPI tool flow, to create customized designs for 256 different ERI quartet classes, based on their orbitals. To maximize data reuse, all intermediates are buffered in FPGA on-chip memory with customized layout. The pre-calculation of intermediates also helps to overcome data dependencies caused by multi-dimensional recurrence
relations. The involved loop structures are partially or even fully unrolled for high throughput of FPGA kernels. Furthermore, a lossy compression algorithm utilizing arbitrary bitwidth integers is integrated in the FPGA kernels. To our
best knowledge, this is the first work on ERI computation on FPGAs that supports more than just the single most basic quartet class. Also, the integration of ERI computation and compression it a novelty that is not even covered by CPU or GPU libraries so far.
Our evaluation shows that using 16-bit integer for the ERI compression, the fastest FPGA kernels exceed the performance of 10 GERIS ($10 \times 10^9$ ERIs per second) on one Intel Stratix 10 GX 2800 FPGA, with maximum absolute errors around $10^{-7}$ - $10^{-5}$ Hartree. The measured throughput can be accurately explained by a performance model. The FPGA kernels deployed on 2 FPGAs outperform similar computations using the widely used libint reference on a two-socket server with 40 Xeon Gold 6148 CPU cores of the same process technology by factors up to 6.0x and on a new two-socket server with 128 EPYC 7713 CPU cores by up to 1.9x.}},
  author       = {{Wu, Xin and Kenter, Tobias and Schade, Robert and Kühne, Thomas and Plessl, Christian}},
  booktitle    = {{2023 IEEE 31st Annual International Symposium on Field-Programmable Custom Computing Machines (FCCM)}},
  pages        = {{162--173}},
  title        = {{{Computing and Compressing Electron Repulsion Integrals on FPGAs}}},
  doi          = {{10.1109/FCCM57271.2023.00026}},
  year         = {{2023}},
}

@article{45361,
  abstract     = {{<jats:p> The non-orthogonal local submatrix method applied to electronic structure–based molecular dynamics simulations is shown to exceed 1.1 EFLOP/s in FP16/FP32-mixed floating-point arithmetic when using 4400 NVIDIA A100 GPUs of the Perlmutter system. This is enabled by a modification of the original method that pushes the sustained fraction of the peak performance to about 80%. Example calculations are performed for SARS-CoV-2 spike proteins with up to 83 million atoms. </jats:p>}},
  author       = {{Schade, Robert and Kenter, Tobias and Elgabarty, Hossam and Lass, Michael and Kühne, Thomas and Plessl, Christian}},
  issn         = {{1094-3420}},
  journal      = {{The International Journal of High Performance Computing Applications}},
  keywords     = {{Hardware and Architecture, Theoretical Computer Science, Software}},
  publisher    = {{SAGE Publications}},
  title        = {{{Breaking the exascale barrier for the electronic structure problem in ab-initio molecular dynamics}}},
  doi          = {{10.1177/10943420231177631}},
  year         = {{2023}},
}

@article{55900,
  author       = {{Scharwald, Dennis and Meier, Torsten and Sharapova, Polina}},
  issn         = {{2643-1564}},
  journal      = {{Physical Review Research}},
  number       = {{4}},
  publisher    = {{American Physical Society (APS)}},
  title        = {{{Phase sensitivity of spatially broadband high-gain SU(1,1) interferometers}}},
  doi          = {{10.1103/physrevresearch.5.043158}},
  volume       = {{5}},
  year         = {{2023}},
}

@article{61252,
  abstract     = {{<jats:title>Abstract</jats:title><jats:p>The biexciton‐exciton emission cascade commonly used in quantum‐dot systems to generate polarization entanglement yields photons with intrinsically limited indistinguishability. In the present work, it focuses on the generation of pairs of photons with high degrees of polarization entanglement and simultaneously high indistinguishability. It achieves this goal by selectively reducing the biexciton lifetime with an optical resonator. It demonstrates that a suitably tailored circular Bragg reflector fulfills the requirements of sufficient selective Purcell enhancement of biexciton emission paired with spectrally broad photon extraction and twofold degenerate optical modes. The in‐depth theoretical study combines (i) the optimization of realistic photonic structures solving Maxwell's equations from which model parameters are extracted as input for (ii) microscopic simulations of quantum‐dot cavity excitation dynamics with full access to photon properties. It reports non‐trivial dependencies on system parameters and use the predictive power of the combined theoretical approach to determine the optimal range of Purcell enhancement that maximizes indistinguishability and entanglement to near unity values, here specifically for the telecom C‐band at 1550 nm.</jats:p>}},
  author       = {{Bauch, David and Siebert, Dustin and Jöns, Klaus D. and Förstner, Jens and Schumacher, Stefan}},
  issn         = {{2511-9044}},
  journal      = {{Advanced Quantum Technologies}},
  number       = {{1}},
  publisher    = {{Wiley}},
  title        = {{{On‐Demand Indistinguishable and Entangled Photons Using Tailored Cavity Designs}}},
  doi          = {{10.1002/qute.202300142}},
  volume       = {{7}},
  year         = {{2023}},
}

@article{61266,
  abstract     = {{<jats:p>This review examines the use of continuous-variable spectroscopy techniques for investigating quantum coherence and light-matter interactions in semiconductor systems with ultrafast dynamics. Special emphasis is placed on multichannel homodyne detection as a powerful tool to measure the quantum coherence and the full density matrix of a polariton system. Observations, such as coherence times that exceed the nanosecond scale obtained by monitoring the temporal decay of quantum coherence in a polariton condensate, are discussed. Proof-of-concept experiments and numerical simulations that demonstrate the enhanced resourcefulness of the produced system states for modern quantum protocols are assessed. The combination of tailored resource quantifiers and ultrafast spectroscopy techniques that have recently been demonstrated paves the way for future applications of quantum information technologies.</jats:p>}},
  author       = {{Lüders, Carolin and Barkhausen, Franziska and Pukrop, Matthias and Rozas, Elena and Sperling, Jan and Schumacher, Stefan and Aßmann, Marc}},
  issn         = {{2159-3930}},
  journal      = {{Optical Materials Express}},
  number       = {{11}},
  publisher    = {{Optica Publishing Group}},
  title        = {{{Continuous-variable quantum optics and resource theory for ultrafast semiconductor spectroscopy [Invited]}}},
  doi          = {{10.1364/ome.497006}},
  volume       = {{13}},
  year         = {{2023}},
}

@article{61264,
  author       = {{Yu, Yueyang and Dong, Chuan-Ding and Binder, Rolf and Schumacher, Stefan and Ning, Cun-Zheng}},
  issn         = {{1936-0851}},
  journal      = {{ACS Nano}},
  number       = {{5}},
  pages        = {{4230--4238}},
  publisher    = {{American Chemical Society (ACS)}},
  title        = {{{Strain-Induced Indirect-to-Direct Bandgap Transition, Photoluminescence Enhancement, and Linewidth Reduction in Bilayer MoTe<sub>2</sub>}}},
  doi          = {{10.1021/acsnano.2c01665}},
  volume       = {{17}},
  year         = {{2023}},
}

@article{61267,
  abstract     = {{<jats:p>Dynamics-induced interchain charge transfer in a polymer aggregate in stack configuration can be understood by single-oligomer polaron energy.</jats:p>}},
  author       = {{Bauch, Fabian and Dong, Chuan-Ding and Schumacher, Stefan}},
  issn         = {{2050-7526}},
  journal      = {{Journal of Materials Chemistry C}},
  number       = {{38}},
  pages        = {{12992--12998}},
  publisher    = {{Royal Society of Chemistry (RSC)}},
  title        = {{{Dynamics-induced charge transfer in semiconducting conjugated polymers}}},
  doi          = {{10.1039/d3tc02263c}},
  volume       = {{11}},
  year         = {{2023}},
}

