@article{15851,
  author       = {{Ma, Xuekai and Kartashov, Yaroslav Y and Gao, Tingge and Schumacher, Stefan}},
  issn         = {{1367-2630}},
  journal      = {{New Journal of Physics}},
  title        = {{{Controllable high-speed polariton waves in a PT-symmetric lattice}}},
  doi          = {{10.1088/1367-2630/ab5a9b}},
  volume       = {{21}},
  year         = {{2019}},
}

@unpublished{13340,
  abstract     = {{Spontaneous formation of transverse patterns is ubiquitous in nonlinear
dynamical systems of all kinds. An aspect of particular interest is the active
control of such patterns. In nonlinear optical systems this can be used for
all-optical switching with transistor-like performance, for example realized
with polaritons in a planar quantum-well semiconductor microcavity. Here we
focus on a specific configuration which takes advantage of the intricate
polarization dependencies in the interacting optically driven polariton system.
Besides detailed numerical simulations of the coupled light-field exciton
dynamics, in the present paper we focus on the derivation of a simplified
population competition model giving detailed insight into the underlying
mechanisms from a nonlinear dynamical systems perspective. We show that such a
model takes the form of a generalized Lotka-Volterra system for two competing
populations explicitly including a source term that enables external control.
We present a comprehensive analysis both of the existence and stability of
stationary states in the parameter space spanned by spatial anisotropy and
external control strength. We also construct phase boundaries in non-trivial
regions and characterize emerging bifurcations. The population competition
model reproduces all key features of the switching observed in full numerical
simulations of the rather complex semiconductor system and at the same time is
simple enough for a fully analytical understanding of the system dynamics.}},
  author       = {{Pukrop, Matthias and Schumacher, Stefan}},
  booktitle    = {{arXiv:1903.12534}},
  title        = {{{Externally Controlled Lotka-Volterra Dynamics in a Linearly Polarized  Polariton Fluid}}},
  year         = {{2019}},
}

@unpublished{13347,
  abstract     = {{<jats:p>&lt;div&gt;
			&lt;div&gt;
				&lt;div&gt;
					&lt;p&gt;Molecular doping in conjugated polymers is a crucial process for their application in organic
photovoltaics and optoelectronics. In the present work we theoretically investigate p-type molecu-
lar doping in a series of (poly[2,6-(4,4-bis(2-ethylhexyl)-4H-cyclopenta[2,1-b;3,4-b”]dithiophene)-alt-
4,7-(2,1,3-benzothiadiazole)] (PCPDT-BT) conjugated oligomers with different lengths and three
widely-used dopants with different electron affinities, namely F4TCNQ, F6TCNNQ, and CN6-CP.
We study in detail the molecular geometry of possible oligomer-dopant complexes and its influence
on the doping mechanisms and electronic system properties. We find that the mechanisms of dop-
ing and charge transfer observed sensitively depend on the specific geometry of the oligomer-dopant
complexes. For a given complex different geometries may exist, some of which show transfer of
an entire electron from the oligomer chain onto the dopant molecule resulting in an integer-charge
transfer complex, leaving the system in a ground state with broken spin symmetry. In other ge-
ometries merely hybridization of oligomer and dopant frontier orbitals occurs with partial charge
transfer but spin-symmetric ground state. Considering the resulting electronic density of states both
cases may well contribute to an increased electrical conductivity of corresponding film samples while
the underlying physical mechanisms are entirely different.
&lt;/p&gt;
				&lt;/div&gt;
			&lt;/div&gt;
		&lt;/div&gt;</jats:p>}},
  author       = {{Dong, Chuan-Ding and Schumacher, Stefan}},
  title        = {{{Molecular Doping of PCPDT-BT Copolymers: Comparison of Molecular Complexes with and Without Integer Charge Transfer}}},
  year         = {{2019}},
}

@article{13343,
  author       = {{Vollbrecht, Joachim and Wiebeler, Christian and Bock, Harald and Schumacher, Stefan and Kitzerow, Heinz-Siegfried}},
  issn         = {{1932-7447}},
  journal      = {{The Journal of Physical Chemistry C}},
  number       = {{7}},
  pages        = {{4483--4492}},
  title        = {{{Curved Polar Dibenzocoronene Esters and Imides versus Their Planar Centrosymmetric Homologs: Photophysical and Optoelectronic Analysis}}},
  doi          = {{10.1021/acs.jpcc.8b10730}},
  volume       = {{123}},
  year         = {{2019}},
}

@article{20,
  abstract     = {{Approximate computing has shown to provide new ways to improve performance
and power consumption of error-resilient applications. While many of these
applications can be found in image processing, data classification or machine
learning, we demonstrate its suitability to a problem from scientific
computing. Utilizing the self-correcting behavior of iterative algorithms, we
show that approximate computing can be applied to the calculation of inverse
matrix p-th roots which are required in many applications in scientific
computing. Results show great opportunities to reduce the computational effort
and bandwidth required for the execution of the discussed algorithm, especially
when targeting special accelerator hardware.}},
  author       = {{Lass, Michael and Kühne, Thomas and Plessl, Christian}},
  issn         = {{1943-0671}},
  journal      = {{Embedded Systems Letters}},
  number       = {{2}},
  pages        = {{ 33--36}},
  publisher    = {{IEEE}},
  title        = {{{Using Approximate Computing for the Calculation of Inverse Matrix p-th Roots}}},
  doi          = {{10.1109/LES.2017.2760923}},
  volume       = {{10}},
  year         = {{2018}},
}

@inproceedings{22,
  abstract     = {{This paper describes a data structure and a heuristic to plan and map arbitrary resources in complex combinations while applying time dependent constraints. The approach is used in the planning based workload manager OpenCCS at the Paderborn Center for Parallel Computing (PC\(^2\)) to operate heterogeneous clusters with up to 10000 cores. We also show performance results derived from four years of operation.}},
  author       = {{Keller, Axel}},
  booktitle    = {{Proc. Workshop on Job Scheduling Strategies for Parallel Processing (JSSPP)}},
  editor       = {{Klusáček, D. and Cirne, W. and Desai, N.}},
  isbn         = {{978-3-319-77398-8}},
  keywords     = {{Scheduling Planning Mapping Workload management}},
  location     = {{Orlando, FL, USA}},
  pages        = {{132--151}},
  publisher    = {{Springer}},
  title        = {{{A Data Structure for Planning Based Workload Management of Heterogeneous HPC Systems}}},
  doi          = {{10.1007/978-3-319-77398-8_8}},
  volume       = {{10773}},
  year         = {{2018}},
}

@misc{5414,
  author       = {{Filmwala, Tasneem}},
  publisher    = {{Universität Paderborn}},
  title        = {{{Study Effects of Approximation on Conjugate Gradient Algorithm and Accelerate it on FPGA Platform}}},
  year         = {{2018}},
}

@misc{5421,
  author       = {{Gadewar, Onkar}},
  publisher    = {{Universität Paderborn}},
  title        = {{{Programmable Programs? - Designing FPGA Overlay Architectures with OpenCL}}},
  year         = {{2018}},
}

@article{6516,
  author       = {{Mertens, Jan Cedric and Boschmann, Alexander and Schmidt, M. and Plessl, Christian}},
  issn         = {{1369-7072}},
  journal      = {{Sports Engineering}},
  number       = {{4}},
  pages        = {{441--451}},
  publisher    = {{Springer Nature}},
  title        = {{{Sprint diagnostic with GPS and inertial sensor fusion}}},
  doi          = {{10.1007/s12283-018-0291-0}},
  volume       = {{21}},
  year         = {{2018}},
}

@misc{5417,
  abstract     = {{Molecular Dynamic (MD) simulations are computationally intensive and accelerating them using specialized hardware is a topic of investigation in many studies. One of the routines in the critical path of MD simulations is the three-dimensional Fast Fourier Transformation (FFT3d). The potential in accelerating FFT3d using hardware is usually bound by bandwidth and memory. Therefore, designing a high throughput solution for an FPGA that overcomes this problem is challenging.
In this thesis, the feasibility of offloading FFT3d computations to FPGA implemented using OpenCL is investigated. In order to mask the latency in memory access, an FFT3d that overlaps computation with communication is designed. The implementa- tion of this design is synthesized for the Arria 10 GX 1150 FPGA and evaluated with the FFTW benchmark. Analysis shows a better performance using FPGA over CPU for larger FFT sizes, with the 643 FFT showing a 70% improvement in runtime using FPGAs.
This FFT3d design is integrated with CP2K to explore the potential in accelerating molecular dynamic simulations. Evaluation of CP2K simulations using FPGA shows a 41% improvement in runtime in FFT3d computations over CPU for larger FFT3d designs.}},
  author       = {{Ramaswami, Arjun}},
  keywords     = {{FFT: FPGA, CP2K, OpenCL}},
  publisher    = {{Universität Paderborn}},
  title        = {{{Accelerating Molecular Dynamic Simulations by Offloading Fast Fourier Transformations to FPGA}}},
  year         = {{2018}},
}

@inproceedings{1588,
  abstract     = {{The exploration of FPGAs as accelerators for scientific simulations has so far mostly been focused on small kernels of methods working on regular data structures, for example in the form of stencil computations for finite difference methods. In computational sciences, often more advanced methods are employed that promise better stability, convergence, locality and scaling. Unstructured meshes are shown to be more effective and more accurate, compared to regular grids, in representing computation domains of various shapes. Using unstructured meshes, the discontinuous Galerkin method preserves the ability to perform explicit local update operations for simulations in the time domain. In this work, we investigate FPGAs as target platform for an implementation of the nodal discontinuous Galerkin method to find time-domain solutions of Maxwell's equations in an unstructured mesh. When maximizing data reuse and fitting constant coefficients into suitably partitioned on-chip memory, high computational intensity allows us to implement and feed wide data paths with hundreds of floating point operators. By decoupling off-chip memory accesses from the computations, high memory bandwidth can be sustained, even for the irregular access pattern required by parts of the application. Using the Intel/Altera OpenCL SDK for FPGAs, we present different implementation variants for different polynomial orders of the method. In different phases of the algorithm, either computational or bandwidth limits of the Arria 10 platform are almost reached, thus outperforming a highly multithreaded CPU implementation by around 2x.}},
  author       = {{Kenter, Tobias and Mahale, Gopinath and Alhaddad, Samer and Grynko, Yevgen and Schmitt, Christian and Afzal, Ayesha and Hannig, Frank and Förstner, Jens and Plessl, Christian}},
  booktitle    = {{Proc. Int. Symp. on Field-Programmable Custom Computing Machines (FCCM)}},
  keywords     = {{tet_topic_hpc}},
  publisher    = {{IEEE}},
  title        = {{{OpenCL-based FPGA Design to Accelerate the Nodal Discontinuous Galerkin Method for Unstructured Meshes}}},
  doi          = {{10.1109/FCCM.2018.00037}},
  year         = {{2018}},
}

@inproceedings{1590,
  abstract     = {{We present the submatrix method, a highly parallelizable method for the approximate calculation of inverse p-th roots of large sparse symmetric matrices which are required in different scientific applications. Following the idea of Approximate Computing, we allow imprecision in the final result in order to utilize the sparsity of the input matrix and to allow massively parallel execution. For an n x n matrix, the proposed algorithm allows to distribute the calculations over n nodes with only little communication overhead. The result matrix exhibits the same sparsity pattern as the input matrix, allowing for efficient reuse of allocated data structures.

We evaluate the algorithm with respect to the error that it introduces into calculated results, as well as its performance and scalability. We demonstrate that the error is relatively limited for well-conditioned matrices and that results are still valuable for error-resilient applications like preconditioning even for ill-conditioned matrices. We discuss the execution time and scaling of the algorithm on a theoretical level and present a distributed implementation of the algorithm using MPI and OpenMP. We demonstrate the scalability of this implementation by running it on a high-performance compute cluster comprised of 1024 CPU cores, showing a speedup of 665x compared to single-threaded execution.}},
  author       = {{Lass, Michael and Mohr, Stephan and Wiebeler, Hendrik and Kühne, Thomas and Plessl, Christian}},
  booktitle    = {{Proc. Platform for Advanced Scientific Computing (PASC) Conference}},
  isbn         = {{978-1-4503-5891-0/18/07}},
  keywords     = {{approximate computing, linear algebra, matrix inversion, matrix p-th roots, numeric algorithm, parallel computing}},
  location     = {{Basel, Switzerland}},
  publisher    = {{ACM}},
  title        = {{{A Massively Parallel Algorithm for the Approximate Calculation of Inverse p-th Roots of Large Sparse Matrices}}},
  doi          = {{10.1145/3218176.3218231}},
  year         = {{2018}},
}

@inproceedings{1204,
  author       = {{Riebler, Heinrich and Vaz, Gavin Francis and Kenter, Tobias and Plessl, Christian}},
  booktitle    = {{Proc. ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP)}},
  isbn         = {{9781450349826}},
  keywords     = {{htrop}},
  publisher    = {{ACM}},
  title        = {{{Automated Code Acceleration Targeting Heterogeneous OpenCL Devices}}},
  doi          = {{10.1145/3178487.3178534}},
  year         = {{2018}},
}

@article{13409,
  author       = {{Biktagirov, Timur and Schmidt, Wolf Gero and Gerstmann, Uwe}},
  issn         = {{2469-9950}},
  journal      = {{Physical Review B}},
  number       = {{11}},
  title        = {{{Calculation of spin-spin zero-field splitting within periodic boundary conditions: Towards all-electron accuracy}}},
  doi          = {{10.1103/physrevb.97.115135}},
  volume       = {{97}},
  year         = {{2018}},
}

@article{13407,
  abstract     = {{<p>A study of structural evolution upon photoinduced charge transfer in a dicopper complex with biologically relevant sulfur coordination.</p>}},
  author       = {{Naumova, Maria and Khakhulin, Dmitry and Rebarz, Mateusz and Rohrmüller, Martin and Dicke, Benjamin and Biednov, Mykola and Britz, Alexander and Espinoza, Shirly and Grimm-Lebsanft, Benjamin and Kloz, Miroslav and Kretzschmar, Norman and Neuba, Adam and Ortmeyer, Jochen and Schoch, Roland and Andreasson, Jakob and Bauer, Matthias and Bressler, Christian and Schmidt, Wolf Gero and Henkel, Gerald and Rübhausen, Michael}},
  issn         = {{1463-9076}},
  journal      = {{Physical Chemistry Chemical Physics}},
  pages        = {{6274--6286}},
  title        = {{{Structural dynamics upon photoexcitation-induced charge transfer in a dicopper(i)–disulfide complex}}},
  doi          = {{10.1039/c7cp04880g}},
  year         = {{2018}},
}

@article{13408,
  author       = {{Lichtenstein, T. and Mamiyev, Z. and Braun, C. and Sanna, S. and Schmidt, Wolf Gero and Tegenkamp, C. and Pfnür, H.}},
  issn         = {{2469-9950}},
  journal      = {{Physical Review B}},
  number       = {{16}},
  title        = {{{Probing quasi-one-dimensional band structures by plasmon spectroscopy}}},
  doi          = {{10.1103/physrevb.97.165421}},
  volume       = {{97}},
  year         = {{2018}},
}

@article{13411,
  author       = {{Halbig, B. and Liebhaber, M. and Bass, U. and Geurts, J. and Speiser, E. and Räthel, J. and Chandola, S. and Esser, N. and Krenz, Marvin and Neufeld, Sergej and Schmidt, Wolf Gero and Sanna, S.}},
  issn         = {{2469-9950}},
  journal      = {{Physical Review B}},
  number       = {{3}},
  title        = {{{Vibrational properties of the Au-(3×3)/Si(111) surface reconstruction}}},
  doi          = {{10.1103/physrevb.97.035412}},
  volume       = {{97}},
  year         = {{2018}},
}

@article{13413,
  author       = {{Seino, Kaori and Sanna, Simone and Schmidt, Wolf Gero}},
  issn         = {{0039-6028}},
  journal      = {{Surface Science}},
  pages        = {{101--104}},
  title        = {{{Temperature stabilizes rough Au/Ge(001) surface reconstructions}}},
  doi          = {{10.1016/j.susc.2017.10.005}},
  volume       = {{667}},
  year         = {{2018}},
}

@article{13430,
  author       = {{Lichtenstein, T. and Mamiyev, Z. and Braun, Christian and Sanna, S. and Schmidt, Wolf Gero and Tegenkamp, C. and Pfnür, H.}},
  issn         = {{2469-9950}},
  journal      = {{Physical Review B}},
  number       = {{16}},
  title        = {{{Probing quasi-one-dimensional band structures by plasmon spectroscopy}}},
  doi          = {{10.1103/physrevb.97.165421}},
  volume       = {{97}},
  year         = {{2018}},
}

@article{13348,
  author       = {{Luk, Samuel M. H. and Lewandowski, P. and Kwong, N. H. and Baudin, E. and Lafont, O. and Tignon, J. and Leung, P. T. and Chan, Ch. K. P. and Babilon, M. and Schumacher, Stefan and Binder, R.}},
  issn         = {{0740-3224}},
  journal      = {{Journal of the Optical Society of America B}},
  number       = {{1}},
  title        = {{{Theory of optically controlled anisotropic polariton transport in semiconductor double microcavities}}},
  doi          = {{10.1364/josab.35.000146}},
  volume       = {{35}},
  year         = {{2018}},
}

