[{"type":"conference","keyword":["electron repulsion integrals","quantum chemistry","atomistic simulation","overlay architecture","fpga acceleration"],"department":[{"_id":"27"},{"_id":"518"}],"date_created":"2026-02-06T06:43:22Z","abstract":[{"text":"The computation of highly contracted electron repulsion integrals (ERIs) is essential to achieve quantum accuracy in atomistic simulations based on quantum mechanics. Its growing computational demands make energy efficiency a critical concern. Recent studies demonstrate FPGAs’ superior performance and energy efficiency for computing primitive ERIs, but the computation of highly contracted ERIs introduces significant algorithmic complexity and new design challenges for FPGA acceleration.In this work, we present SORCERI, the first streaming overlay acceleration for highly contracted ERI computations on FPGAs. SORCERI introduces a novel streaming Rys computing unit to calculate roots and weights of Rys polynomials on-chip, and a streaming contraction unit for the contraction of primitive ERIs. This shifts the design bottleneck from limited CPU-FPGA communication bandwidth to available FPGA computation resources. To address practical deployment challenges for a large number of quartet classes, we design three streaming overlays, together with an efficient memory transpose optimization, to cover the 21 most commonly used quartet classes in realistic atomistic simulations. To address the new computation constraints, we use flexible calculation stages with a free-running streaming architecture to achieve high DSP utilization and good timing closure.Experiments demonstrate that SORCERI achieves an average 5.96x, 1.99x, and 1.16x better performance per watt than libint on a 64-core AMD EPYC 7713 CPU, libintx on an Nvidia A40 GPU, and SERI, the prior best-performing FPGA design for primitive ERIs. Furthermore, SORCERI reaches a peak throughput of 44.11 GERIS (109 ERIs per second) that is 1.52x, 1.13x, and 1.93x greater than libint, libintx and SERI, respectively. SORCERI will be released soon at https://github.com/SFU-HiAccel/SORCERI.","lang":"eng"}],"publication":"Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA '26)","doi":"10.1145/3748173.3779198","main_file_link":[{"url":"https://dl.acm.org/doi/10.1145/3748173.3779198"}],"language":[{"iso":"eng"}],"date_updated":"2026-02-09T09:16:32Z","publication_status":"published","year":"2026","title":"SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry","publication_identifier":{"isbn":["9798400720796"]},"author":[{"full_name":"Stachura, Philip","last_name":"Stachura","first_name":"Philip"},{"id":"77439","full_name":"Wu, Xin","last_name":"Wu","first_name":"Xin"},{"id":"16153","orcid":"0000-0001-5728-9982","first_name":"Christian","last_name":"Plessl","full_name":"Plessl, Christian"},{"last_name":"Fang","first_name":"Zhenman","full_name":"Fang, Zhenman"}],"place":"New York, NY, USA","project":[{"name":"Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"}],"citation":{"short":"P. Stachura, X. Wu, C. Plessl, Z. Fang, in: Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26), Association for Computing Machinery, New York, NY, USA, 2026, pp. 224–234.","chicago":"Stachura, Philip, Xin Wu, Christian Plessl, and Zhenman Fang. “SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry.” In <i>Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)</i>, 224–34. New York, NY, USA: Association for Computing Machinery, 2026. <a href=\"https://doi.org/10.1145/3748173.3779198\">https://doi.org/10.1145/3748173.3779198</a>.","ieee":"P. Stachura, X. Wu, C. Plessl, and Z. Fang, “SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry,” in <i>Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)</i>, 2026, pp. 224–234, doi: <a href=\"https://doi.org/10.1145/3748173.3779198\">10.1145/3748173.3779198</a>.","apa":"Stachura, P., Wu, X., Plessl, C., &#38; Fang, Z. (2026). SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry. <i>Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)</i>, 224–234. <a href=\"https://doi.org/10.1145/3748173.3779198\">https://doi.org/10.1145/3748173.3779198</a>","bibtex":"@inproceedings{Stachura_Wu_Plessl_Fang_2026, place={New York, NY, USA}, title={SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry}, DOI={<a href=\"https://doi.org/10.1145/3748173.3779198\">10.1145/3748173.3779198</a>}, booktitle={Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)}, publisher={Association for Computing Machinery}, author={Stachura, Philip and Wu, Xin and Plessl, Christian and Fang, Zhenman}, year={2026}, pages={224–234} }","ama":"Stachura P, Wu X, Plessl C, Fang Z. SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry. In: <i>Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)</i>. Association for Computing Machinery; 2026:224-234. doi:<a href=\"https://doi.org/10.1145/3748173.3779198\">10.1145/3748173.3779198</a>","mla":"Stachura, Philip, et al. “SORCERI: Streaming Overlay Acceleration for Highly Contracted Electron Repulsion Integral Computations in Quantum Chemistry.” <i>Proceedings of the 2026 ACM/SIGDA International Symposium on Field Programmable Gate Arrays (FPGA ’26)</i>, Association for Computing Machinery, 2026, pp. 224–34, doi:<a href=\"https://doi.org/10.1145/3748173.3779198\">10.1145/3748173.3779198</a>."},"user_id":"77439","page":"224-234","_id":"63890","publisher":"Association for Computing Machinery","status":"public"},{"year":"2019","title":"Zynq-based acceleration of robust high density myoelectric signal processing","author":[{"first_name":"Alexander","last_name":"Boschmann","full_name":"Boschmann, Alexander"},{"full_name":"Agne, Andreas","first_name":"Andreas","last_name":"Agne"},{"full_name":"Thombansen, Georg","first_name":"Georg","last_name":"Thombansen"},{"id":"49051","first_name":"Linus Matthias","last_name":"Witschen","full_name":"Witschen, Linus Matthias"},{"full_name":"Kraus, Florian","first_name":"Florian","last_name":"Kraus"},{"full_name":"Platzner, Marco","last_name":"Platzner","first_name":"Marco","id":"398"}],"publication_identifier":{"issn":["0743-7315"]},"publication_status":"published","date_updated":"2022-01-06T06:51:13Z","intvolume":"       123","language":[{"iso":"eng"}],"doi":"10.1016/j.jpdc.2018.07.004","publication":"Journal of Parallel and Distributed Computing","abstract":[{"lang":"eng","text":"Advances in electromyographic (EMG) sensor technology and machine learning algorithms have led to an increased research effort into high density EMG-based pattern recognition methods for prosthesis control. With the goal set on an autonomous multi-movement prosthesis capable of performing training and classification of an amputee’s EMG signals, the focus of this paper lies in the acceleration of the embedded signal processing chain. We present two Xilinx Zynq-based architectures for accelerating two inherently different high density EMG-based control algorithms. The first hardware accelerated design achieves speed-ups of up to 4.8 over the software-only solution, allowing for a processing delay lower than the sample period of 1 ms. The second system achieved a speed-up of 5.5 over the software-only version and operates at a still satisfactory low processing delay of up to 15 ms while providing a higher reliability and robustness against electrode shift and noisy channels."}],"date_created":"2019-07-12T13:13:55Z","keyword":["High density electromyography","FPGA acceleration","Medical signal processing","Pattern recognition","Prosthetics"],"type":"journal_article","department":[{"_id":"78"}],"status":"public","page":"77-89","publisher":"Elsevier","_id":"11950","user_id":"398","volume":123,"citation":{"short":"A. Boschmann, A. Agne, G. Thombansen, L.M. Witschen, F. Kraus, M. Platzner, Journal of Parallel and Distributed Computing 123 (2019) 77–89.","chicago":"Boschmann, Alexander, Andreas Agne, Georg Thombansen, Linus Matthias Witschen, Florian Kraus, and Marco Platzner. “Zynq-Based Acceleration of Robust High Density Myoelectric Signal Processing.” <i>Journal of Parallel and Distributed Computing</i> 123 (2019): 77–89. <a href=\"https://doi.org/10.1016/j.jpdc.2018.07.004\">https://doi.org/10.1016/j.jpdc.2018.07.004</a>.","ieee":"A. Boschmann, A. Agne, G. Thombansen, L. M. Witschen, F. Kraus, and M. Platzner, “Zynq-based acceleration of robust high density myoelectric signal processing,” <i>Journal of Parallel and Distributed Computing</i>, vol. 123, pp. 77–89, 2019.","apa":"Boschmann, A., Agne, A., Thombansen, G., Witschen, L. M., Kraus, F., &#38; Platzner, M. (2019). Zynq-based acceleration of robust high density myoelectric signal processing. <i>Journal of Parallel and Distributed Computing</i>, <i>123</i>, 77–89. <a href=\"https://doi.org/10.1016/j.jpdc.2018.07.004\">https://doi.org/10.1016/j.jpdc.2018.07.004</a>","bibtex":"@article{Boschmann_Agne_Thombansen_Witschen_Kraus_Platzner_2019, title={Zynq-based acceleration of robust high density myoelectric signal processing}, volume={123}, DOI={<a href=\"https://doi.org/10.1016/j.jpdc.2018.07.004\">10.1016/j.jpdc.2018.07.004</a>}, journal={Journal of Parallel and Distributed Computing}, publisher={Elsevier}, author={Boschmann, Alexander and Agne, Andreas and Thombansen, Georg and Witschen, Linus Matthias and Kraus, Florian and Platzner, Marco}, year={2019}, pages={77–89} }","ama":"Boschmann A, Agne A, Thombansen G, Witschen LM, Kraus F, Platzner M. Zynq-based acceleration of robust high density myoelectric signal processing. <i>Journal of Parallel and Distributed Computing</i>. 2019;123:77-89. doi:<a href=\"https://doi.org/10.1016/j.jpdc.2018.07.004\">10.1016/j.jpdc.2018.07.004</a>","mla":"Boschmann, Alexander, et al. “Zynq-Based Acceleration of Robust High Density Myoelectric Signal Processing.” <i>Journal of Parallel and Distributed Computing</i>, vol. 123, Elsevier, 2019, pp. 77–89, doi:<a href=\"https://doi.org/10.1016/j.jpdc.2018.07.004\">10.1016/j.jpdc.2018.07.004</a>."}},{"language":[{"iso":"eng"}],"_id":"9784","page":"277-280","doi":"10.1109/ULTSYM.2012.0068","user_id":"55222","author":[{"last_name":"Hunstig","first_name":"Matthias","full_name":"Hunstig, Matthias"},{"last_name":"Hemsel","first_name":"Tobias","full_name":"Hemsel, Tobias"},{"last_name":"Sextro","first_name":"Walter","full_name":"Sextro, Walter"}],"publication_identifier":{"issn":["1948-5719"]},"title":"An efficient simulation technique for high-frequency piezoelectric inertia motors","year":"2012","status":"public","date_updated":"2022-01-06T07:04:20Z","date_created":"2019-05-13T13:20:17Z","department":[{"_id":"151"}],"type":"conference","keyword":["friction","ultrasonic motors","Coulomb friction model","efficient simulation technique","friction contact","high-frequency piezoelectric inertia motor","motor characteristics prediction","numerical simulation","slip-slip mode","stick-slip mode","time-step simulation","ultrasonic inertia motor","Acceleration","Acoustics","Actuators","Computational modeling","Friction","Numerical models","Steady-state"],"citation":{"mla":"Hunstig, Matthias, et al. “An Efficient Simulation Technique for High-Frequency Piezoelectric Inertia Motors.” <i>Ultrasonics Symposium (IUS), 2012 IEEE International</i>, 2012, pp. 277–80, doi:<a href=\"https://doi.org/10.1109/ULTSYM.2012.0068\">10.1109/ULTSYM.2012.0068</a>.","bibtex":"@inproceedings{Hunstig_Hemsel_Sextro_2012, title={An efficient simulation technique for high-frequency piezoelectric inertia motors}, DOI={<a href=\"https://doi.org/10.1109/ULTSYM.2012.0068\">10.1109/ULTSYM.2012.0068</a>}, booktitle={Ultrasonics Symposium (IUS), 2012 IEEE International}, author={Hunstig, Matthias and Hemsel, Tobias and Sextro, Walter}, year={2012}, pages={277–280} }","ama":"Hunstig M, Hemsel T, Sextro W. An efficient simulation technique for high-frequency piezoelectric inertia motors. In: <i>Ultrasonics Symposium (IUS), 2012 IEEE International</i>. ; 2012:277-280. doi:<a href=\"https://doi.org/10.1109/ULTSYM.2012.0068\">10.1109/ULTSYM.2012.0068</a>","ieee":"M. Hunstig, T. Hemsel, and W. Sextro, “An efficient simulation technique for high-frequency piezoelectric inertia motors,” in <i>Ultrasonics Symposium (IUS), 2012 IEEE International</i>, 2012, pp. 277–280.","apa":"Hunstig, M., Hemsel, T., &#38; Sextro, W. (2012). An efficient simulation technique for high-frequency piezoelectric inertia motors. In <i>Ultrasonics Symposium (IUS), 2012 IEEE International</i> (pp. 277–280). <a href=\"https://doi.org/10.1109/ULTSYM.2012.0068\">https://doi.org/10.1109/ULTSYM.2012.0068</a>","chicago":"Hunstig, Matthias, Tobias Hemsel, and Walter Sextro. “An Efficient Simulation Technique for High-Frequency Piezoelectric Inertia Motors.” In <i>Ultrasonics Symposium (IUS), 2012 IEEE International</i>, 277–80, 2012. <a href=\"https://doi.org/10.1109/ULTSYM.2012.0068\">https://doi.org/10.1109/ULTSYM.2012.0068</a>.","short":"M. Hunstig, T. Hemsel, W. Sextro, in: Ultrasonics Symposium (IUS), 2012 IEEE International, 2012, pp. 277–280."},"publication":"Ultrasonics Symposium (IUS), 2012 IEEE International","abstract":[{"lang":"eng","text":"Piezoelectric inertia motors use the inertia of a body to drive it by means of a friction contact in a series of small steps. These motors can operate in ``stick-slip'' or ``slip-slip'' mode, with the fundamental frequency of the driving signal ranging from several Hertz to more than 100 kHz. To predict the motor characteristics, a Coulomb friction model is sufficient in many cases, but numerical simulation requires microscopic time steps. This contribution proposes a much faster simulation technique using one evaluation per period of the excitation signal. The proposed technique produces results very close to those of timestep simulation for ultrasonics inertia motors and allows direct determination of the steady-state velocity of an inertia motor from the motion profile of the driving part. Thus it is a useful simulation technique which can be applied in both analysis and design of inertia motors, especially for parameter studies and optimisation."}],"quality_controlled":"1"},{"abstract":[{"lang":"eng","text":"IP-XACT is a well accepted standard for the exchange of IP components at Electronic System and Register Transfer Level. Still, the creation and manipulation of these descriptions at the XML level can be time-consuming and error-prone. In this paper, we show that the UML can be consistently applied as an efficient and comprehensible frontend for IP-XACT-based IP description and integration. For this, we present an IP-XACT UML profile that enables UML-based descriptions covering the same information as a corresponding IP-XACT description. This enables the automated generation of IP-XACT component and design descriptions from respective UML models. In particular, it also allows the integration of existing IPs with UML. To illustrate our approach, we present an application example based on the IBM PowerPC Evaluation Kit."}],"citation":{"ieee":"T. Schattkowsky, T. Xie, and W. Müller, “A UML Frontend for IP-XACT-based IP Management,” presented at the Design, Automation &#38; Test in Europe Conference &#38; Exhibition, 2009, doi: <a href=\"https://doi.org/10.1109/DATE.2009.5090664\">10.1109/DATE.2009.5090664</a>.","apa":"Schattkowsky, T., Xie, T., &#38; Müller, W. (2009). A UML Frontend for IP-XACT-based IP Management. <i>Proceedings of DATE’09</i>. Design, Automation &#38; Test in Europe Conference &#38; Exhibition. <a href=\"https://doi.org/10.1109/DATE.2009.5090664\">https://doi.org/10.1109/DATE.2009.5090664</a>","short":"T. Schattkowsky, T. Xie, W. Müller, in: Proceedings of DATE’09, IEEE, Nice, France, 2009.","chicago":"Schattkowsky, Tim, Tao Xie, and Wolfgang Müller. “A UML Frontend for IP-XACT-Based IP Management.” In <i>Proceedings of DATE’09</i>. Nice, France: IEEE, 2009. <a href=\"https://doi.org/10.1109/DATE.2009.5090664\">https://doi.org/10.1109/DATE.2009.5090664</a>.","mla":"Schattkowsky, Tim, et al. “A UML Frontend for IP-XACT-Based IP Management.” <i>Proceedings of DATE’09</i>, IEEE, 2009, doi:<a href=\"https://doi.org/10.1109/DATE.2009.5090664\">10.1109/DATE.2009.5090664</a>.","bibtex":"@inproceedings{Schattkowsky_Xie_Müller_2009, place={Nice, France}, title={A UML Frontend for IP-XACT-based IP Management}, DOI={<a href=\"https://doi.org/10.1109/DATE.2009.5090664\">10.1109/DATE.2009.5090664</a>}, booktitle={Proceedings of DATE’09}, publisher={IEEE}, author={Schattkowsky, Tim and Xie, Tao and Müller, Wolfgang}, year={2009} }","ama":"Schattkowsky T, Xie T, Müller W. A UML Frontend for IP-XACT-based IP Management. In: <i>Proceedings of DATE’09</i>. IEEE; 2009. doi:<a href=\"https://doi.org/10.1109/DATE.2009.5090664\">10.1109/DATE.2009.5090664</a>"},"publication":"Proceedings of DATE'09","department":[{"_id":"672"}],"keyword":["Unified modeling language","XML","Power system modeling","Application software","Master-slave","Power system management","Acceleration","Scattering","Software engineering","Software standards"],"type":"conference","date_created":"2023-01-17T11:54:02Z","place":"Nice, France","date_updated":"2023-01-17T11:54:07Z","publication_identifier":{"isbn":["978-1-4244-3781-8"]},"author":[{"first_name":"Tim","last_name":"Schattkowsky","full_name":"Schattkowsky, Tim"},{"first_name":"Tao","last_name":"Xie","full_name":"Xie, Tao"},{"full_name":"Müller, Wolfgang","first_name":"Wolfgang","last_name":"Müller","id":"16243"}],"conference":{"name":"Design, Automation & Test in Europe Conference & Exhibition"},"status":"public","year":"2009","title":"A UML Frontend for IP-XACT-based IP Management","user_id":"5786","doi":"10.1109/DATE.2009.5090664","_id":"37067","language":[{"iso":"eng"}],"publisher":"IEEE"},{"_id":"2420","publisher":"Kluwer Academic Publishers","page":"109-129","volume":26,"user_id":"398","status":"public","citation":{"bibtex":"@article{Plessl_Platzner_2003, title={Instance-Specific Accelerators for Minimum Covering}, volume={26}, DOI={<a href=\"https://doi.org/10.1023/a:1024443416592\">10.1023/a:1024443416592</a>}, number={2}, journal={Journal of Supercomputing}, publisher={Kluwer Academic Publishers}, author={Plessl, Christian and Platzner, Marco}, year={2003}, pages={109–129} }","ama":"Plessl C, Platzner M. Instance-Specific Accelerators for Minimum Covering. <i>Journal of Supercomputing</i>. 2003;26(2):109-129. doi:<a href=\"https://doi.org/10.1023/a:1024443416592\">10.1023/a:1024443416592</a>","mla":"Plessl, Christian, and Marco Platzner. “Instance-Specific Accelerators for Minimum Covering.” <i>Journal of Supercomputing</i>, vol. 26, no. 2, Kluwer Academic Publishers, 2003, pp. 109–29, doi:<a href=\"https://doi.org/10.1023/a:1024443416592\">10.1023/a:1024443416592</a>.","short":"C. Plessl, M. Platzner, Journal of Supercomputing 26 (2003) 109–129.","chicago":"Plessl, Christian, and Marco Platzner. “Instance-Specific Accelerators for Minimum Covering.” <i>Journal of Supercomputing</i> 26, no. 2 (2003): 109–29. <a href=\"https://doi.org/10.1023/a:1024443416592\">https://doi.org/10.1023/a:1024443416592</a>.","ieee":"C. Plessl and M. Platzner, “Instance-Specific Accelerators for Minimum Covering,” <i>Journal of Supercomputing</i>, vol. 26, no. 2, pp. 109–129, 2003.","apa":"Plessl, C., &#38; Platzner, M. (2003). Instance-Specific Accelerators for Minimum Covering. <i>Journal of Supercomputing</i>, <i>26</i>(2), 109–129. <a href=\"https://doi.org/10.1023/a:1024443416592\">https://doi.org/10.1023/a:1024443416592</a>"},"language":[{"iso":"eng"}],"doi":"10.1023/a:1024443416592","author":[{"full_name":"Plessl, Christian","first_name":"Christian","last_name":"Plessl","orcid":"0000-0001-5728-9982","id":"16153"},{"id":"398","full_name":"Platzner, Marco","last_name":"Platzner","first_name":"Marco"}],"publication_identifier":{"issn":["0920-8542"]},"year":"2003","title":"Instance-Specific Accelerators for Minimum Covering","intvolume":"        26","date_updated":"2022-01-06T06:56:10Z","date_created":"2018-04-17T15:10:00Z","department":[{"_id":"518"},{"_id":"78"}],"type":"journal_article","keyword":["reconfigurable computing","instance-specific acceleration","minimum covering"],"publication":"Journal of Supercomputing","issue":"2","abstract":[{"lang":"eng","text":" This paper presents the acceleration of minimum-cost covering problems by instance-specific hardware. First, we formulate the minimum-cost covering problem and discuss a branch \\& bound algorithm to solve it. Then we describe instance-specific hardware architectures that implement branch \\& bound in 3-valued logic and use reduction techniques similar to those found in software solvers. We further present prototypical accelerator implementations and a corresponding design tool flow. Our experiments reveal significant raw speedups up to five orders of magnitude for a set of smaller unate covering problems. Provided that hardware compilation times can be reduced, we conclude that instance-specific acceleration of hard minimum-cost covering problems will lead to substantial overall speedups. "}],"extern":"1"}]
