@article{46469,
  abstract     = {{We show how to learn discrete field theories from observational data of fields on a space-time lattice. For this, we train a neural network model of a discrete Lagrangian density such that the discrete Euler--Lagrange equations are consistent with the given training data. We, thus, obtain a structure-preserving machine learning architecture. Lagrangian densities are not uniquely defined by the solutions of a field theory. We introduce a technique to derive regularisers for the training process which optimise numerical regularity of the discrete field theory. Minimisation of the regularisers guarantees that close to the training data the discrete field theory behaves robust and efficient when used in numerical simulations. Further, we show how to identify structurally simple solutions of the underlying continuous field theory such as travelling waves. This is possible even when travelling waves are not present in the training data. This is compared to data-driven model order reduction based approaches, which struggle to identify suitable latent spaces containing structurally simple solutions when these are not present in the training data. Ideas are demonstrated on examples based on the wave equation and the Schrödinger equation. }},
  author       = {{Offen, Christian and Ober-Blöbaum, Sina}},
  issn         = {{1054-1500}},
  journal      = {{Chaos}},
  number       = {{1}},
  publisher    = {{AIP Publishing}},
  title        = {{{Learning of discrete models of variational PDEs from data}}},
  doi          = {{10.1063/5.0172287}},
  volume       = {{34}},
  year         = {{2024}},
}

@unpublished{55159,
  abstract     = {{We introduce a method based on Gaussian process regression to identify discrete variational principles from observed solutions of a field theory. The method is based on the data-based identification of a discrete Lagrangian density. It is a geometric machine learning technique in the sense that the variational structure of the true field theory is reflected in the data-driven model by design. We provide a rigorous convergence statement of the method. The proof circumvents challenges posed by the ambiguity of discrete Lagrangian densities in the inverse problem of variational calculus.
Moreover, our method can be used to quantify model uncertainty in the equations of motions and any linear observable of the discrete field theory. This is illustrated on the example of the discrete wave equation and Schrödinger equation.
The article constitutes an extension of our previous article  arXiv:2404.19626 for the data-driven identification of (discrete) Lagrangians for variational dynamics from an ode setting to the setting of discrete pdes.}},
  author       = {{Offen, Christian}},
  keywords     = {{System identification, inverse problem of variational calculus, Gaussian process, Lagrangian learning, physics informed machine learning, geometry aware learning}},
  pages        = {{28}},
  title        = {{{Machine learning of discrete field theories with guaranteed convergence and uncertainty quantification}}},
  year         = {{2024}},
}

@article{55989,
  abstract     = {{Phased arrays are vital in communication systems and have received significant interest in the field of optoelectronics and photonics, enabling a wide range of applications such as LiDAR, holography, wireless communication, etc. In this work, we present a blazed grating antenna that is optimized to have upward radiation efficiency as high as 80% with a compact footprint of 3.5 μm × 2 μm at an operational wavelength of 1.55 μm. Our numerical investigations demonstrate that this antenna in a 64 × 64 phased array configuration is capable of producing desired far-field radiation patterns. Additionally, our antenna possesses a low side lobe level of -9.7 dB and a negligible reflection efficiency of under 1%, making it an attractive candidate for integrated optical phased arrays.}},
  author       = {{Farheen, Henna and Joshi, Suraj and Scheytt, J. Christoph and Myroshnychenko, Viktor and Förstner, Jens}},
  issn         = {{2515-7647}},
  journal      = {{Journal of Physics: Photonics}},
  keywords     = {{tet_topic_opticalantenna}},
  pages        = {{045010}},
  publisher    = {{IOP Publishing}},
  title        = {{{An efficient compact blazed grating antenna for optical phased arrays}}},
  doi          = {{10.1088/2515-7647/ad6ed4}},
  volume       = {{6}},
  year         = {{2024}},
}

@inproceedings{56194,
  author       = {{Afsahnoudeh, Reza and Riese, Julia and Kenig, Eugeny Y.}},
  booktitle    = {{World Congress on Mechanical, Chemical, and Material Engineering}},
  issn         = {{2369-8136}},
  location     = {{Barcelona}},
  publisher    = {{Avestia Publishing}},
  title        = {{{A Numerical Analysis of Thermo-Hydraulic Performance of Pillow-Plate Heat Exchangers with Ellipsoidal Secondary Structures}}},
  doi          = {{10.11159/htff24.145}},
  year         = {{2024}},
}

@inproceedings{56605,
  author       = {{Opdenhövel, Jan-Oliver and Alt, Christoph and Plessl, Christian and Kenter, Tobias}},
  booktitle    = {{2024 34th International Conference on Field-Programmable Logic and Applications (FPL)}},
  publisher    = {{IEEE}},
  title        = {{{StencilStream: A SYCL-based Stencil Simulation Framework Targeting FPGAs}}},
  doi          = {{10.1109/fpl64840.2024.00023}},
  year         = {{2024}},
}

@inproceedings{56609,
  abstract     = {{The computation of electron repulsion integrals (ERIs) is a key component for quantum chemical methods. The intensive computation and bandwidth demand for ERI evaluation presents a significant challenge for quantum-mechanics-based atomistic simulations with hybrid density functional theory: due to the tens of trillions of ERI computations in each time step, practical applications are usually limited to thousands of atoms. In this work, we propose SERI, a high-throughput streaming accelerator for ERI computation on HBM-based FPGAs. In contrast to prior buffer-based designs, SERI proposes a novel streaming architecture to address the on-chip buffer limitation and the floorplanning challenge, and leverages the high-bandwidth memory to overcome the bandwidth bottleneck in prior designs. Moreover, to meet the varying computation, bandwidth, and floorplanning requirements between the 55 canonical quartet classes in ERI calculation, we design an automation tool, together with an accurate performance model, to automatically customize the architecture and floorplanning strategy for each canonical quartet class to maximize their throughput. Our performance evaluation on the AMD/Xilinx Alveo U280 FPGA board shows that, SERI achieves an average speedup of 9.80 x over the previous best-performing FPGA design, a 3.21x speedup over a 64-core AMD EPYC 7713 CPU, and a 15.64x speedup over an Nvidia A40 GPU. It reaches a peak throughput of 23.8 GERIS ($10^9$ ERIs per second) on one Alveo U280 FPGA. SERI will be released soon at https://github.com/SFU-HiAccel/SERI.}},
  author       = {{Stachura, Philip and Li, Guanyu and Wu, Xin and Plessl, Christian and Fang, Zhenman}},
  booktitle    = {{2024 34th International Conference on Field-Programmable Logic and Applications (FPL)}},
  pages        = {{60--68}},
  publisher    = {{IEEE}},
  title        = {{{SERI: High-Throughput Streaming Acceleration of Electron Repulsion Integral Computation in Quantum Chemistry using HBM-based FPGAs}}},
  doi          = {{10.1109/fpl64840.2024.00018}},
  year         = {{2024}},
}

@article{56678,
  abstract     = {{<jats:title>Abstract</jats:title><jats:p>The surface area of atoms and molecules plays a crucial role in shaping many physiochemical properties of materials. Despite its fundamental importance, precisely defining atomic and molecular surfaces has long been a puzzle. Among the available definitions, a straightforward and elegant approach by Bader describes a molecular surface as an iso-density surface beyond which the electron density drops below a certain cut-off. However, so far neither this theory nor a decisive value for the density cut-off have been amenable to experimental verification due to the limitations of conventional experimental methods. In the present study, we employ a state-of-the-art experimental method based on the recently developed concept of thermodynamically effective (TE) surfaces to tackle this longstanding problem. By studying a set of 104 molecules, a close to perfect agreement between quantum chemical evaluations of iso-density surfaces contoured at a cut-off density of 0.0016 a.u. and experimental results obtained via thermodynamic phase change data is demonstrated, with a mean unsigned percentage deviation of 1.6% and a correlation coefficient of 0.995. Accordingly, we suggest the iso-density surface contoured at an electron density value of 0.0016 a.u. as a representation of the surface of atoms and molecules.</jats:p>}},
  author       = {{Alibakhshi, Amin and Schäfer, Lars V.}},
  issn         = {{2041-1723}},
  journal      = {{Nature Communications}},
  number       = {{1}},
  publisher    = {{Springer Science and Business Media LLC}},
  title        = {{{Electron iso-density surfaces provide a thermodynamically consistent representation of atomic and molecular surfaces}}},
  doi          = {{10.1038/s41467-024-50408-8}},
  volume       = {{15}},
  year         = {{2024}},
}

@article{56679,
  author       = {{Alibakhshi, Amin and Schäfer, Lars V.}},
  issn         = {{1089-5639}},
  journal      = {{The Journal of Physical Chemistry A}},
  number       = {{32}},
  pages        = {{6819--6823}},
  publisher    = {{American Chemical Society (ACS)}},
  title        = {{{On the Theoretical Quantification of Radii of Atoms in Molecules}}},
  doi          = {{10.1021/acs.jpca.4c04529}},
  volume       = {{128}},
  year         = {{2024}},
}

@article{56778,
  author       = {{Bernemann, Sören Antonius and Maćkowiak, J.F. and Maćkowiak, J. and Kenig, Eugeny}},
  issn         = {{0009-2509}},
  journal      = {{Chemical Engineering Science}},
  publisher    = {{Elsevier BV}},
  title        = {{{Computer aided flow investigation of liquid agricultural wastes}}},
  doi          = {{10.1016/j.ces.2024.120639}},
  volume       = {{300}},
  year         = {{2024}},
}

@article{52958,
  author       = {{Boeddeker, Christoph and Subramanian, Aswin Shanmugam and Wichern, Gordon and Haeb-Umbach, Reinhold and Le Roux, Jonathan}},
  issn         = {{2329-9290}},
  journal      = {{IEEE/ACM Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{Electrical and Electronic Engineering, Acoustics and Ultrasonics, Computer Science (miscellaneous), Computational Mathematics}},
  pages        = {{1185--1197}},
  publisher    = {{Institute of Electrical and Electronics Engineers (IEEE)}},
  title        = {{{TS-SEP: Joint Diarization and Separation Conditioned on Estimated Speaker Embeddings}}},
  doi          = {{10.1109/taslp.2024.3350887}},
  volume       = {{32}},
  year         = {{2024}},
}

@inproceedings{55638,
  abstract     = {{<jats:p>Abstract. Traditionally, joints are cylindrical and rotationally symmetric. In the present study, non-rotationally symmetric joints are used for joining steel and Glass mat-reinforced thermoplastic sheets (GMT). In addition, the study also analyzes the impact of non-rotational symmetric joint rotation on the load-bearing capacity. Single lap joint specimens were fabricated using the In-Mold assembly technique for joining steel sheets with GMT. Tensile shear tests were performed on different orientations of the joint geometry, and it was observed that changing the joint orientation influences the load-bearing capacity. The joints are constitutively modeled using beam elements and the influence of joint rotation on load distribution is examined through a static simulation study. </jats:p>}},
  author       = {{Devulapally, Deekshith Reddy and Martin, Sven and Tröster, Thomas}},
  booktitle    = {{Materials Research Proceedings}},
  issn         = {{2474-395X}},
  publisher    = {{Materials Research Forum LLC}},
  title        = {{{Non-rotationally symmetric joints – Mechanisms and load bearing capacity}}},
  doi          = {{10.21741/9781644903131-183}},
  year         = {{2024}},
}

@article{61251,
  abstract     = {{<jats:p>We theoretically investigate strategies for the deterministic creation of trains of time-bin entangled photons using an individual quantum emitter described by a Λ-type electronic system. We explicitly demonstrate the theoretical generation of linear cluster states with substantial numbers of entangled photonic qubits in full microscopic numerical simulations. The underlying scheme is based on the manipulation of ground state coherences through precise optical driving. One important finding is that the most easily accessible quality metrics, the achievable rotation fidelities, fall short in assessing the actual quantum correlations of the emitted photons in the face of losses. To address this, we explicitly calculate stabilizer generator expectation values as a superior gauge for the quantum properties of the generated many-photon state. With widespread applicability in other emitter and excitation–emission schemes also, our work lays the conceptual foundations for an in-depth practical analysis of time-bin entanglement based on full numerical simulations with predictive capabilities for realistic systems and setups, including losses and imperfections. The specific results shown in the present work illustrate that with controlled minimization of losses and realistic system parameters for quantum-dot type systems, useful linear cluster states of significant lengths can be generated in the calculations, discussing the possibility of scalability for quantum information processing endeavors.</jats:p>}},
  author       = {{Bauch, David and Köcher, Nikolas and Heinisch, Nils and Schumacher, Stefan}},
  issn         = {{2835-0103}},
  journal      = {{APL Quantum}},
  number       = {{3}},
  publisher    = {{AIP Publishing}},
  title        = {{{Time-bin entanglement in the deterministic generation of linear photonic cluster states}}},
  doi          = {{10.1063/5.0214197}},
  volume       = {{1}},
  year         = {{2024}},
}

@article{61253,
  abstract     = {{<jats:p>In the SUPER scheme (Swing-UP of the quantum EmitteR population), excitation of a quantum emitter is achieved with two off-resonant, red-detuned laser pulses. This allows the generation of high-quality single photons without the need of complex laser stray light suppression or careful spectral filtering. In the present work, we extend this promising method to quantum emitters, specifically semiconductor quantum dots, inside a resonant optical cavity. A significant advantage of the SUPER scheme is identified in that it eliminates re-excitation of the quantum emitter by suppressing photon emission during the excitation cycle. This, in turn, leads to almost ideal single-photon purity, overcoming a major factor typically limiting the quality of photons generated with quantum emitters in high-quality cavities. We further find that for cavity-mediated biexciton emission of degenerate photon pairs, the SUPER scheme leads to near-perfect biexciton initialization with very high values of polarization entanglement of emitted photon pairs.</jats:p>
          <jats:sec>
            <jats:title/>
            <jats:supplementary-material>
              <jats:permissions>
                <jats:copyright-statement>Published by the American Physical Society</jats:copyright-statement>
                <jats:copyright-year>2024</jats:copyright-year>
              </jats:permissions>
            </jats:supplementary-material>
          </jats:sec>}},
  author       = {{Heinisch, Nils and Köcher, Nikolas and Bauch, David and Schumacher, Stefan}},
  issn         = {{2643-1564}},
  journal      = {{Physical Review Research}},
  number       = {{1}},
  publisher    = {{American Physical Society (APS)}},
  title        = {{{Swing-up dynamics in quantum emitter cavity systems: Near ideal single photons and entangled photon pairs}}},
  doi          = {{10.1103/physrevresearch.6.l012017}},
  volume       = {{6}},
  year         = {{2024}},
}

@article{61255,
  abstract     = {{<jats:title>Abstract</jats:title>
               <jats:p>Topological states have been widely investigated in different types of systems and lattices. In the present work, we report on topological edge states in double-wave (DW) chains, which can be described by a generalized Aubry-André-Harper (AAH) model. For the specific system of a driven-dissipative exciton polariton system we show that in such potential chains, different types of edge states can form. For resonant optical excitation, we further find that the optical nonlinearity leads to a multistability of different edge states. This includes topologically protected edge states evolved directly from individual linear eigenstates as well as additional edge states that originate from nonlinearity-induced localization of bulk states. Extending the system into two dimensions (2D) by stacking horizontal DW chains in the vertical direction, we also create 2D multi-wave lattices. In such 2D lattices multiple Su–Schrieffer–Heeger (SSH) chains appear along the vertical direction. The combination of DW chains in the horizonal and SSH chains in the vertical direction then results in the formation of higher-order topological insulator corner states. Multistable corner states emerge in the nonlinear regime.</jats:p>}},
  author       = {{Schneider, Tobias and Gao, Wenlong and Zentgraf, Thomas and Schumacher, Stefan and Ma, Xuekai}},
  issn         = {{2192-8614}},
  journal      = {{Nanophotonics}},
  number       = {{4}},
  pages        = {{509--518}},
  publisher    = {{Walter de Gruyter GmbH}},
  title        = {{{Topological edge and corner states in coupled wave lattices in nonlinear polariton condensates}}},
  doi          = {{10.1515/nanoph-2023-0556}},
  volume       = {{13}},
  year         = {{2024}},
}

@article{61257,
  abstract     = {{<jats:p>Exceptional points (EPs), with their intriguing spectral topology, have attracted considerable attention in a broad range of physical systems, with potential sensing applications driving much of the present research in this field. Here, we investigate spectral topology and EPs in systems with significant nonlinearity, exemplified by a nonequilibrium exciton-polariton condensate. With the possibility to control loss and gain and nonlinearity by optical means, this system allows for a comprehensive analysis of the interplay of nonlinearities (Kerr type and saturable gain) and non-Hermiticity. Not only do we find that EPs can be intentionally shifted in parameter space by the saturable gain, but we also observe intriguing rotations and intersections of Riemann surfaces and find nonlinearity-enhanced sensing capabilities. With this, our results illustrate the potential of tailoring spectral topology and related phenomena in non-Hermitian systems by nonlinearity.</jats:p>
          <jats:sec>
            <jats:title/>
            <jats:supplementary-material>
              <jats:permissions>
                <jats:copyright-statement>Published by the American Physical Society</jats:copyright-statement>
                <jats:copyright-year>2024</jats:copyright-year>
              </jats:permissions>
            </jats:supplementary-material>
          </jats:sec>}},
  author       = {{Wingenbach, Jan and Schumacher, Stefan and Ma, Xuekai}},
  issn         = {{2643-1564}},
  journal      = {{Physical Review Research}},
  number       = {{1}},
  publisher    = {{American Physical Society (APS)}},
  title        = {{{Manipulating spectral topology and exceptional points by nonlinearity in non-Hermitian polariton systems}}},
  doi          = {{10.1103/physrevresearch.6.013148}},
  volume       = {{6}},
  year         = {{2024}},
}

@article{61259,
  author       = {{Bauch, Fabian and Dong, Chuan-Ding and Schumacher, Stefan}},
  issn         = {{1932-7447}},
  journal      = {{The Journal of Physical Chemistry C}},
  number       = {{8}},
  pages        = {{3525--3532}},
  publisher    = {{American Chemical Society (ACS)}},
  title        = {{{Dynamics of Electron–Hole Coulomb Attractive Energy and Dipole Moment of Hot Excitons in Donor–Acceptor Polymers}}},
  doi          = {{10.1021/acs.jpcc.3c07513}},
  volume       = {{128}},
  year         = {{2024}},
}

@article{61263,
  abstract     = {{<jats:p>Charge transfer mechanism in the deprotonation-induced n-type doping of PCBM.</jats:p>}},
  author       = {{Dong, Chuan-Ding and Bauch, Fabian and Hu, Yuanyuan and Schumacher, Stefan}},
  issn         = {{1463-9076}},
  journal      = {{Physical Chemistry Chemical Physics}},
  number       = {{5}},
  pages        = {{4194--4199}},
  publisher    = {{Royal Society of Chemistry (RSC)}},
  title        = {{{Charge transfer in superbase n-type doping of PCBM induced by deprotonation}}},
  doi          = {{10.1039/d3cp05105f}},
  volume       = {{26}},
  year         = {{2024}},
}

@article{61357,
  author       = {{Krenz, Marvin and Sanna, Simone and Gerstmann, Uwe and Schmidt, Wolf Gero}},
  issn         = {{1932-7447}},
  journal      = {{The Journal of Physical Chemistry C}},
  number       = {{41}},
  pages        = {{17774--17778}},
  publisher    = {{American Chemical Society (ACS)}},
  title        = {{{Understanding and Improving Triplet Exciton Transfer in Sensitized Silicon Solar Cells}}},
  doi          = {{10.1021/acs.jpcc.4c05446}},
  volume       = {{128}},
  year         = {{2024}},
}

@inbook{62067,
  abstract     = {{Most FPGA boards in the HPC domain are well-suited for parallel scaling because of the direct integration of versatile and high-throughput network ports. However, the utilization of their network capabilities is often challenging and error-prone because the whole network stack and communication patterns have to be implemented and managed on the FPGAs. Also, this approach conceptually involves a trade-off between the performance potential of improved communication and the impact of resource consumption for communication infrastructure, since the utilized resources on the FPGAs could otherwise be used for computations. In this work, we investigate this trade-off, firstly, by using synthetic benchmarks to evaluate the different configuration options of the communication framework ACCL and their impact on communication latency and throughput. Finally, we use our findings to implement a shallow water simulation whose scalability heavily depends on low-latency communication. With a suitable configuration of ACCL, good scaling behavior can be shown to all 48 FPGAs installed in the system. Overall, the results show that the availability of inter-FPGA communication frameworks as well as the configurability of framework and network stack are crucial to achieve the best application performance with low latency communication.}},
  author       = {{Meyer, Marius and Kenter, Tobias and Petrica, Lucian and O’Brien, Kenneth and Blott, Michaela and Plessl, Christian}},
  booktitle    = {{Lecture Notes in Computer Science}},
  isbn         = {{9783031697654}},
  issn         = {{0302-9743}},
  publisher    = {{Springer Nature Switzerland}},
  title        = {{{Optimizing Communication for Latency Sensitive HPC Applications on up to 48 FPGAs Using ACCL}}},
  doi          = {{10.1007/978-3-031-69766-1_9}},
  year         = {{2024}},
}

@article{56604,
  abstract     = {{This manuscript makes the claim of having computed the 9th Dedekind number, D(9). This was done by accelerating the core operation of the process with an efficient FPGA design that outperforms an optimized 64-core CPU reference by 95x. The FPGA execution was parallelized on the Noctua 2 supercomputer at Paderborn University. The resulting value for D(9) is 286386577668298411128469151667598498812366. This value can be verified in two steps. We have made the data file containing the 490 M results available, each of which can be verified separately on CPU, and the whole file sums to our proposed value. The paper explains the mathematical approach in the first part, before putting the focus on a deep dive into the FPGA accelerator implementation followed by a performance analysis. The FPGA implementation was done in Register-Transfer Level using a dual-clock architecture and shows how we achieved an impressive FMax of 450 MHz on the targeted Stratix 10 GX 2,800 FPGAs. The total compute time used was 47,000 FPGA hours.}},
  author       = {{Van Hirtum, Lennart and De Causmaecker, Patrick and Goemaere, Jens and Kenter, Tobias and Riebler, Heinrich and Lass, Michael and Plessl, Christian}},
  issn         = {{1936-7406}},
  journal      = {{ACM Transactions on Reconfigurable Technology and Systems}},
  number       = {{3}},
  pages        = {{1--28}},
  publisher    = {{Association for Computing Machinery (ACM)}},
  title        = {{{A Computation of the Ninth Dedekind Number Using FPGA Supercomputing}}},
  doi          = {{10.1145/3674147}},
  volume       = {{17}},
  year         = {{2024}},
}

