@phdthesis{10624,
  abstract     = {{The use of heterogeneous computing resources, such as graphics processing units or other specialized co-processors, has become widespread in recent years because of their performance and energy efficiency advantages. Operating system approaches that are limited to optimizing CPU usage are no longer sufficient for the efficient utilization of systems that comprise diverse resource types.

Enabling task preemption on these architectures and migration of tasks between different resource types at run-time is not only key to improving the performance and energy consumption but also to enabling automatic scheduling methods for heterogeneous compute nodes.

This thesis proposes novel techniques for run-time management of heterogeneous resources and enabling tasks to migrate between diverse hardware. It provides fundamental work towards future operating systems by discussing implications, limitations, and chances of the heterogeneity and introducing solutions for energy- and performance-efficient run-time systems. Scheduling methods to utilize heterogeneous systems by the use of a centralized scheduler are presented that show benefits over existing approaches in varying case studies.}},
  author       = {{Beisel, Tobias}},
  isbn         = {{978-3-8325-4155-2}},
  pages        = {{183}},
  publisher    = {{Logos Verlag Berlin GmbH}},
  title        = {{{Management and Scheduling of Accelerators for Heterogeneous High-Performance Computing}}},
  year         = {{2015}},
}

@misc{10668,
  author       = {{Hangmann, Hendrik}},
  publisher    = {{Paderborn University}},
  title        = {{{Evolution of Heat Flow Prediction Models for FPGA Devices}}},
  year         = {{2015}},
}

@misc{10671,
  author       = {{Haupt, Christian}},
  publisher    = {{Paderborn University}},
  title        = {{{Computer Vision basierte Klassifikation von HD EMG Signalen}}},
  year         = {{2015}},
}

@inproceedings{10673,
  author       = {{Ho, Nam and Ahmed, Abdullah Fathi and Kaufmann, Paul and Platzner, Marco}},
  booktitle    = {{Proc. NASA/ESA Conf. Adaptive Hardware and Systems (AHS)}},
  keywords     = {{cache storage, field programmable gate arrays, multiprocessing systems, parallel architectures, reconfigurable architectures, FPGA, dynamic reconfiguration, evolvable cache mapping, many-core architecture, memory-to-cache address mapping function, microarchitectural optimization, multicore architecture, nature-inspired optimization, parallelization degrees, processor, reconfigurable cache mapping, reconfigurable computing, Field programmable gate arrays, Software, Tuning}},
  pages        = {{1--7}},
  title        = {{{Microarchitectural optimization by means of reconfigurable and evolvable cache mappings}}},
  doi          = {{10.1109/AHS.2015.7231178}},
  year         = {{2015}},
}

@inproceedings{10693,
  author       = {{Kaufmann, Paul and Shen, Cong}},
  booktitle    = {{Genetic and Evolutionary Computation (GECCO)}},
  pages        = {{409--416}},
  publisher    = {{ACM}},
  title        = {{{Generator Start-up Sequences Optimization for Network Restoration Using Genetic Algorithm and Simulated Annealing}}},
  year         = {{2015}},
}

@inproceedings{10711,
  author       = {{Meisner, Sebastian and Platzner, Marco}},
  booktitle    = {{Field Programmable Technology (FPT), 2015 International Conference on}},
  pages        = {{212--215}},
  title        = {{{Comparison of thread signatures for error detection in hybrid multi-cores}}},
  doi          = {{10.1109/FPT.2015.7393153}},
  year         = {{2015}},
}

@misc{10714,
  author       = {{Meißner, Roland}},
  publisher    = {{Universität Paderborn}},
  title        = {{{Konzept und Implementation einer Benutzeroberfläche zur Generierung virtueller FPGAs}}},
  year         = {{2015}},
}

@misc{10726,
  author       = {{Posewsky, Thorbjörn}},
  publisher    = {{Paderborn University}},
  title        = {{{Acceleration of Artificial Neural Networks on a Zynq Platform}}},
  year         = {{2015}},
}

@book{10757,
  author       = {{M. Mora, Antonio and Squillero, Giovanni and Agapitos, Alexandros and Burelli, Paolo and S. Bush, William and Cagnoni, Stefano and Cotta, Carlos and De Falco, Ivanoe and Della Cioppa, Antonio and Divina, Federico and Eiben, A.E. and I. Esparcia-Alc{\'a}zar, Anna and Fern{\'a}ndez de Vega, Francisco and Glette, Kyrre and Haasdijk, Evert and Ignacio Hidalgo, J. and Kampouridis, Michael and Kaufmann, Paul and Mavrovouniotis, Michalis and Thanh Nguyen, Trung and Schaefer, Robert and Sim, Kevin and Tarantino, Ernesto and Urquhart, Neil and Zhang (editors), Mengjie}},
  publisher    = {{Springer}},
  title        = {{{Applications of Evolutionary Computation - 18th European Conference, EvoApplications}}},
  volume       = {{9028}},
  year         = {{2015}},
}

@inproceedings{10765,
  author       = {{H.W. Leong, Philip and Amano, Hideharu and Anderson, Jason and Bertels, Koen and M.P. Cardoso, Jo\~ao and Diessel, Oliver and Gogniat, Guy and Hutton, Mike and Lee, JunKyu and Luk, Wayne and Lysaght, Patrick and Platzner, Marco and K. Prasanna, Viktor and Rissa, Tero and Silvano, Cristina and So, Hayden and Wang, Yu}},
  booktitle    = {{Proceedings of the 25th International Conference on Field Programmable Logic and Applications (FPL)}},
  pages        = {{1--3}},
  publisher    = {{Imperial College}},
  title        = {{{Significant papers from the first 25 years of the FPL conference}}},
  doi          = {{10.1109/FPL.2015.7293747}},
  year         = {{2015}},
}

@inproceedings{10767,
  author       = {{Ghribi, Ines and Ben Abdallah, Riadh and Khalgui, Mohamed and Platzner, Marco}},
  booktitle    = {{Proceedings of the 29th European Simulation and Modelling Conference (ESM)}},
  title        = {{{New Codesign Solutions for Modelling and Partitioning of Probabilistic Reconfigurable Embedded Software}}},
  year         = {{2015}},
}

@article{10770,
  author       = {{Ghasemzadeh Mohammadi, Hassan and Gaillardon, Pierre-Emmanuel and De Micheli, Giovanni}},
  journal      = {{IEEE Transactions on Nanotechnology}},
  number       = {{6}},
  pages        = {{1117--1126}},
  publisher    = {{IEEE}},
  title        = {{{From Defect Analysis to Gate-Level Fault Modeling of Controllable-Polarity Silicon Nanowires}}},
  doi          = {{10.1109/TNANO.2015.2482359}},
  volume       = {{14}},
  year         = {{2015}},
}

@inproceedings{10771,
  author       = {{Ghasemzadeh Mohammadi, Hassan and Gaillardon, Pierre-Emmanuel and Zhang, Jian and De Micheli, Giovanni and Sanchez, Eduardo and Reorda, Matteo Sonza}},
  booktitle    = {{2015 IEEE Computer Society Annual Symposium on VLSI}},
  pages        = {{491--496}},
  publisher    = {{IEEE}},
  title        = {{{On the design of a fault tolerant ripple-carry adder with controllable-polarity transistors}}},
  doi          = {{10.1109/ISVLSI.2015.13}},
  year         = {{2015}},
}

@inproceedings{10772,
  author       = {{Ghasemzadeh Mohammadi, Hassan and Gaillardon, Pierre-Emmanuel and De Micheli, Giovanni}},
  booktitle    = {{Proceedings of the 2015 Design, Automation & Test in Europe Conference \& Exhibition}},
  pages        = {{453--458}},
  publisher    = {{EDA Consortium}},
  title        = {{{Fault modeling in controllable polarity silicon nanowire circuits}}},
  doi          = {{10.7873/DATE.2015.0428}},
  year         = {{2015}},
}

@inproceedings{10779,
  author       = {{Guettatfi, Zakarya and Kermia, Omar and Khouas, Abdelhakim}},
  booktitle    = {{25th International Conference on Field Programmable Logic and Applications (FPL)}},
  issn         = {{1946-147X}},
  keywords     = {{embedded systems, field programmable gate arrays, operating systems (computers), scheduling, μC/OS-II, FPGAs, OS foundation, SafeRTOS, Xenomai, chip utilization ration, complex time constraints, embedded systems, hard real-time hardware task allocation, hard real-time hardware task scheduling, hardware-software real-time operating systems, partially reconfigurable field-programmable gate arrays, resource constraints, safety-critical RTOS, Field programmable gate arrays, Hardware, Job shop scheduling, Real-time systems, Shape, Software}},
  publisher    = {{Imperial College}},
  title        = {{{Over effective hard real-time hardware tasks scheduling and allocation}}},
  doi          = {{10.1109/FPL.2015.7293994}},
  year         = {{2015}},
}

@inproceedings{13153,
  author       = {{Graf, Tobias and Platzner, Marco}},
  booktitle    = {{Advances in Computer Games: 14th International Conference, ACG 2015, Leiden, The Netherlands, July 1-3, 2015, Revised Selected Papers}},
  pages        = {{1--11}},
  publisher    = {{Springer International Publishing}},
  title        = {{{Adaptive Playouts in Monte-Carlo Tree Search with Policy-Gradient Reinforcement Learning}}},
  doi          = {{10.1007/978-3-319-27992-3_1}},
  year         = {{2015}},
}

@article{296,
  abstract     = {{FPGAs are known to permit huge gains in performance and efficiency for suitable applications but still require reduced design efforts and shorter development cycles for wider adoption. In this work, we compare the resulting performance of two design concepts that in different ways promise such increased productivity. As common starting point, we employ a kernel-centric design approach, where computational hotspots in an application are identified and individually accelerated on FPGA. By means of a complex stereo matching application, we evaluate two fundamentally different design philosophies and approaches for implementing the required kernels on FPGAs. In the first implementation approach, we designed individually specialized data flow kernels in a spatial programming language for a Maxeler FPGA platform; in the alternative design approach, we target a vector coprocessor with large vector lengths, which is implemented as a form of programmable overlay on the application FPGAs of a Convey HC-1. We assess both approaches in terms of overall system performance, raw kernel performance, and performance relative to invested resources. After compensating for the effects of the underlying hardware platforms, the specialized dataflow kernels on the Maxeler platform are around 3x faster than kernels executing on the Convey vector coprocessor. In our concrete scenario, due to trade-offs between reconfiguration overheads and exposed parallelism, the advantage of specialized dataflow kernels is reduced to around 2.5x.}},
  author       = {{Kenter, Tobias and Schmitz, Henning and Plessl, Christian}},
  journal      = {{International Journal of Reconfigurable Computing (IJRC)}},
  publisher    = {{Hindawi}},
  title        = {{{Exploring Tradeoffs between Specialized Kernels and a Reusable Overlay in a Stereo-Matching Case Study}}},
  doi          = {{10.1155/2015/859425}},
  volume       = {{2015}},
  year         = {{2015}},
}

@inproceedings{303,
  abstract     = {{This paper introduces Binary Acceleration At Runtime(BAAR), an easy-to-use on-the-fly binary acceleration mechanismwhich aims to tackle the problem of enabling existentsoftware to automatically utilize accelerators at runtime. BAARis based on the LLVM Compiler Infrastructure and has aclient-server architecture. The client runs the program to beaccelerated in an environment which allows program analysisand profiling. Program parts which are identified as suitable forthe available accelerator are exported and sent to the server.The server optimizes these program parts for the acceleratorand provides RPC execution for the client. The client transformsits program to utilize accelerated execution on the server foroffloaded program parts. We evaluate our work with a proofof-concept implementation of BAAR that uses an Intel XeonPhi 5110P as the acceleration target and performs automaticoffloading, parallelization and vectorization of suitable programparts. The practicality of BAAR for real-world examples is shownbased on a study of stencil codes. Our results show a speedup ofup to 4 without any developer-provided hints and 5.77 withhints over the same code compiled with the Intel Compiler atoptimization level O2 and running on an Intel Xeon E5-2670machine. Based on our insights gained during implementationand evaluation we outline future directions of research, e.g.,offloading more fine-granular program parts than functions, amore sophisticated communication mechanism or introducing onstack-replacement.}},
  author       = {{Damschen, Marvin and Plessl, Christian}},
  booktitle    = {{Proceedings of the 5th International Workshop on Adaptive Self-tuning Computing Systems (ADAPT)}},
  title        = {{{Easy-to-Use On-The-Fly Binary Program Acceleration on Many-Cores}}},
  year         = {{2015}},
}

@inproceedings{1773,
  author       = {{Schumacher, Jörn and T. Anderson, J. and Borga, A. and Boterenbrood, H. and Chen, H. and Chen, K. and Drake, G. and Francis, D. and Gorini, B. and Lanni, F. and Lehmann-Miotto, Giovanna and Levinson, L. and Narevicius, J. and Plessl, Christian and Roich, A. and Ryu, S. and P. Schreuder, F. and Vandelli, Wainer and Vermeulen, J. and Zhang, J.}},
  booktitle    = {{Proc. Int. Conf. on Distributed Event-Based Systems (DEBS)}},
  publisher    = {{ACM}},
  title        = {{{Improving Packet Processing Performance in the ATLAS FELIX Project – Analysis and Optimization of a Memory-Bounded Algorithm}}},
  doi          = {{10.1145/2675743.2771824}},
  year         = {{2015}},
}

@article{1768,
  author       = {{Plessl, Christian and Platzner, Marco and Schreier, Peter J.}},
  journal      = {{Informatik Spektrum}},
  keywords     = {{approximate computing, survey}},
  number       = {{5}},
  pages        = {{396--399}},
  publisher    = {{Springer}},
  title        = {{{Aktuelles Schlagwort: Approximate Computing}}},
  doi          = {{10.1007/s00287-015-0911-z}},
  year         = {{2015}},
}

