[{"language":[{"iso":"eng"}],"author":[{"id":"34851","last_name":"Ebbers","first_name":"Janek","full_name":"Ebbers, Janek"},{"id":"11213","full_name":"Drude, Lukas","first_name":"Lukas","last_name":"Drude"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"},{"last_name":"Brendel","first_name":"Andreas","full_name":"Brendel, Andreas"},{"last_name":"Kellermann","first_name":"Walter","full_name":"Kellermann, Walter"}],"title":"Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks","year":"2019","date_updated":"2023-11-22T08:29:58Z","date_created":"2020-02-05T10:20:17Z","file":[{"creator":"huesera","date_created":"2020-02-05T10:21:39Z","relation":"main_file","date_updated":"2020-02-05T10:21:39Z","file_name":"CAMSAP_2019_WS_Ebbers_Paper.pdf","access_level":"open_access","file_size":311887,"file_id":"15797","content_type":"application/pdf"}],"department":[{"_id":"54"}],"type":"conference","publication":"CAMSAP 2019, Guadeloupe, West Indies","abstract":[{"lang":"eng","text":"In this paper we consider human daily activity recognition using an acoustic sensor network (ASN) which consists of nodes distributed in a home environment. Assuming that the ASN is permanently recording, the vast majority of recordings is silence. Therefore, we propose to employ a computationally efficient two-stage sound recognition system, consisting of an initial sound activity detection (SAD) and a subsequent sound event classification (SEC), which is only activated once sound activity has been detected. We show how a low-latency activity detector with high temporal resolution can be trained from weak labels with low temporal resolution. We further demonstrate the advantage of using spatial features for the subsequent event classification task."}],"_id":"15796","ddc":["000"],"user_id":"34851","status":"public","has_accepted_license":"1","oa":"1","citation":{"apa":"Ebbers, J., Drude, L., Haeb-Umbach, R., Brendel, A., &#38; Kellermann, W. (2019). Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks. <i>CAMSAP 2019, Guadeloupe, West Indies</i>.","ieee":"J. Ebbers, L. Drude, R. Haeb-Umbach, A. Brendel, and W. Kellermann, “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks,” 2019.","short":"J. Ebbers, L. Drude, R. Haeb-Umbach, A. Brendel, W. Kellermann, in: CAMSAP 2019, Guadeloupe, West Indies, 2019.","chicago":"Ebbers, Janek, Lukas Drude, Reinhold Haeb-Umbach, Andreas Brendel, and Walter Kellermann. “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks.” In <i>CAMSAP 2019, Guadeloupe, West Indies</i>, 2019.","mla":"Ebbers, Janek, et al. “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks.” <i>CAMSAP 2019, Guadeloupe, West Indies</i>, 2019.","ama":"Ebbers J, Drude L, Haeb-Umbach R, Brendel A, Kellermann W. Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks. In: <i>CAMSAP 2019, Guadeloupe, West Indies</i>. ; 2019.","bibtex":"@inproceedings{Ebbers_Drude_Haeb-Umbach_Brendel_Kellermann_2019, title={Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks}, booktitle={CAMSAP 2019, Guadeloupe, West Indies}, author={Ebbers, Janek and Drude, Lukas and Haeb-Umbach, Reinhold and Brendel, Andreas and Kellermann, Walter}, year={2019} }"},"file_date_updated":"2020-02-05T10:21:39Z","project":[{"_id":"52","name":"Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"quality_controlled":"1"},{"date_created":"2020-02-05T10:07:53Z","file":[{"access_level":"open_access","file_size":454600,"file_name":"INTERSPEECH_2019_Ebbers_Paper.pdf","date_updated":"2020-02-05T10:11:40Z","relation":"main_file","content_type":"application/pdf","file_id":"15793","creator":"huesera","date_created":"2020-02-05T10:11:40Z"}],"department":[{"_id":"54"}],"type":"conference","publication":"INTERSPEECH 2019, Graz, Austria","abstract":[{"lang":"eng","text":"In this paper we highlight the privacy risks entailed in deep neural network feature extraction for domestic activity monitoring. We employ the baseline system proposed in the Task 5 of the DCASE 2018 challenge and simulate a feature interception attack by an eavesdropper who wants to perform speaker identification. We then propose to reduce the aforementioned privacy risks by introducing a variational information feature extraction scheme that allows for good activity monitoring performance while at the same time minimizing the information of the feature representation, thus restricting speaker identification attempts. We analyze the resulting model’s composite loss function and the budget scaling factor used to control the balance between the performance of the trusted and attacker tasks. It is empirically demonstrated that the proposed method reduces speaker identification privacy risks without significantly deprecating the performance of domestic activity monitoring tasks."}],"language":[{"iso":"eng"}],"author":[{"first_name":"Alexandru","last_name":"Nelus","full_name":"Nelus, Alexandru"},{"id":"34851","full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"},{"full_name":"Martin, Rainer","first_name":"Rainer","last_name":"Martin"}],"title":"Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification","year":"2019","date_updated":"2023-11-22T08:27:55Z","oa":"1","citation":{"mla":"Nelus, Alexandru, et al. “Privacy-Preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification.” <i>INTERSPEECH 2019, Graz, Austria</i>, 2019.","ama":"Nelus A, Ebbers J, Haeb-Umbach R, Martin R. Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification. In: <i>INTERSPEECH 2019, Graz, Austria</i>. ; 2019.","bibtex":"@inproceedings{Nelus_Ebbers_Haeb-Umbach_Martin_2019, title={Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification}, booktitle={INTERSPEECH 2019, Graz, Austria}, author={Nelus, Alexandru and Ebbers, Janek and Haeb-Umbach, Reinhold and Martin, Rainer}, year={2019} }","apa":"Nelus, A., Ebbers, J., Haeb-Umbach, R., &#38; Martin, R. (2019). Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification. <i>INTERSPEECH 2019, Graz, Austria</i>.","ieee":"A. Nelus, J. Ebbers, R. Haeb-Umbach, and R. Martin, “Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification,” 2019.","chicago":"Nelus, Alexandru, Janek Ebbers, Reinhold Haeb-Umbach, and Rainer Martin. “Privacy-Preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification.” In <i>INTERSPEECH 2019, Graz, Austria</i>, 2019.","short":"A. Nelus, J. Ebbers, R. Haeb-Umbach, R. Martin, in: INTERSPEECH 2019, Graz, Austria, 2019."},"file_date_updated":"2020-02-05T10:11:40Z","quality_controlled":"1","_id":"15792","ddc":["000"],"user_id":"34851","status":"public","has_accepted_license":"1"},{"year":"2018","status":"public","title":"Performance of Mask Based Statistical Beamforming in a Smart Home Scenario","author":[{"id":"9168","last_name":"Heymann","first_name":"Jahn","full_name":"Heymann, Jahn"},{"last_name":"Bacchiani","first_name":"M.","full_name":"Bacchiani, M."},{"last_name":"Sainath","first_name":"T. N.","full_name":"Sainath, T. N."}],"date_updated":"2022-01-06T06:53:26Z","page":"6722-6726","language":[{"iso":"eng"}],"_id":"18107","user_id":"44006","doi":"10.1109/ICASSP.2018.8462372","publication":"2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","citation":{"short":"J. Heymann, M. Bacchiani, T.N. Sainath, in: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 2018, pp. 6722–6726.","ama":"Heymann J, Bacchiani M, Sainath TN. Performance of Mask Based Statistical Beamforming in a Smart Home Scenario. In: <i>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. ; 2018:6722-6726. doi:<a href=\"https://doi.org/10.1109/ICASSP.2018.8462372\">10.1109/ICASSP.2018.8462372</a>","chicago":"Heymann, Jahn, M. Bacchiani, and T. N. Sainath. “Performance of Mask Based Statistical Beamforming in a Smart Home Scenario.” In <i>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, 6722–26, 2018. <a href=\"https://doi.org/10.1109/ICASSP.2018.8462372\">https://doi.org/10.1109/ICASSP.2018.8462372</a>.","bibtex":"@inproceedings{Heymann_Bacchiani_Sainath_2018, title={Performance of Mask Based Statistical Beamforming in a Smart Home Scenario}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2018.8462372\">10.1109/ICASSP.2018.8462372</a>}, booktitle={2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, author={Heymann, Jahn and Bacchiani, M. and Sainath, T. N.}, year={2018}, pages={6722–6726} }","apa":"Heymann, J., Bacchiani, M., &#38; Sainath, T. N. (2018). Performance of Mask Based Statistical Beamforming in a Smart Home Scenario. In <i>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i> (pp. 6722–6726). <a href=\"https://doi.org/10.1109/ICASSP.2018.8462372\">https://doi.org/10.1109/ICASSP.2018.8462372</a>","mla":"Heymann, Jahn, et al. “Performance of Mask Based Statistical Beamforming in a Smart Home Scenario.” <i>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, 2018, pp. 6722–26, doi:<a href=\"https://doi.org/10.1109/ICASSP.2018.8462372\">10.1109/ICASSP.2018.8462372</a>.","ieee":"J. Heymann, M. Bacchiani, and T. N. Sainath, “Performance of Mask Based Statistical Beamforming in a Smart Home Scenario,” in <i>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, 2018, pp. 6722–6726."},"date_created":"2020-08-20T14:20:59Z","type":"conference","department":[{"_id":"54"}]},{"abstract":[{"text":"Acoustic event detection, i.e., the task of assigning a human interpretable label to a segment of audio, has only recently attracted increased interest in the research community. Driven by the DCASE challenges and the availability of large-scale audio datasets, the state-of-the-art has progressed rapidly with deep-learning-based classi- fiers dominating the field. Because several potential use cases favor a realization on distributed sensor nodes, e.g. ambient assisted living applications, habitat monitoring or surveillance, we are concerned with two issues here. Firstly the classification performance of such systems and secondly the computing resources required to achieve a certain performance considering node level feature extraction. In this contribution we look at the balance between the two criteria by employing traditional techniques and different deep learning architectures, including convolutional and recurrent models in the context of real life everyday audio recordings in realistic, however challenging, multisource conditions.","lang":"eng"}],"citation":{"chicago":"Ebbers, Janek, Alexandru Nelus, Rainer Martin, and Reinhold Haeb-Umbach. “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection.” In <i>DAGA 2018, München</i>, 2018.","short":"J. Ebbers, A. Nelus, R. Martin, R. Haeb-Umbach, in: DAGA 2018, München, 2018.","ieee":"J. Ebbers, A. Nelus, R. Martin, and R. Haeb-Umbach, “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection,” in <i>DAGA 2018, München</i>, 2018.","apa":"Ebbers, J., Nelus, A., Martin, R., &#38; Haeb-Umbach, R. (2018). Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection. In <i>DAGA 2018, München</i>.","bibtex":"@inproceedings{Ebbers_Nelus_Martin_Haeb-Umbach_2018, title={Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection}, booktitle={DAGA 2018, München}, author={Ebbers, Janek and Nelus, Alexandru and Martin, Rainer and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Ebbers J, Nelus A, Martin R, Haeb-Umbach R. Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection. In: <i>DAGA 2018, München</i>. ; 2018.","mla":"Ebbers, Janek, et al. “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection.” <i>DAGA 2018, München</i>, 2018."},"publication":"DAGA 2018, München","department":[{"_id":"54"}],"oa":"1","type":"conference","date_created":"2019-07-12T05:27:43Z","date_updated":"2022-01-06T06:51:08Z","author":[{"id":"34851","first_name":"Janek","last_name":"Ebbers","full_name":"Ebbers, Janek"},{"first_name":"Alexandru","last_name":"Nelus","full_name":"Nelus, Alexandru"},{"last_name":"Martin","first_name":"Rainer","full_name":"Martin, Rainer"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"year":"2018","title":"Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection","status":"public","user_id":"44006","_id":"11760","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/Daga_2018_Ebbers_Paper.pdf","open_access":"1"}]},{"date_updated":"2022-01-06T06:51:11Z","author":[{"id":"9168","last_name":"Heymann","first_name":"Jahn","full_name":"Heymann, Jahn"},{"last_name":"Drude","first_name":"Lukas","full_name":"Drude, Lukas","id":"11213"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"},{"first_name":"Keisuke","last_name":"Kinoshita","full_name":"Kinoshita, Keisuke"},{"full_name":"Nakatani, Tomohiro","last_name":"Nakatani","first_name":"Tomohiro"}],"year":"2018","status":"public","title":"Frame-Online DNN-WPE Dereverberation","user_id":"44006","language":[{"iso":"eng"}],"_id":"11835","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/IWAENC_2018_Heymann_Paper.pdf","open_access":"1"}],"related_material":{"link":[{"description":"Poster","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2018/IWAENC_2018_Heymann_Poster.pdf"}]},"abstract":[{"lang":"eng","text":"Signal dereverberation using the weighted prediction error (WPE) method has been proven to be an effective means to raise the accuracy of far-field speech recognition. But in its original formulation, WPE requires multiple iterations over a sufficiently long utterance, rendering it unsuitable for online low-latency applications. Recently, two methods have been proposed to overcome this limitation. One utilizes a neural network to estimate the power spectral density (PSD) of the target signal and works in a block-online fashion. The other method relies on a rather simple PSD estimation which smoothes the observed PSD and utilizes a recursive formulation which enables it to work on a frame-by-frame basis. In this paper, we integrate a deep neural network (DNN) based estimator into the recursive frame-online formulation. We evaluate the performance of the recursive system with different PSD estimators in comparison to the block-online and offline variant on two distinct corpora. The REVERB challenge data, where the signal is mainly deteriorated by reverberation, and a database which combines WSJ and VoiceHome to also consider (directed) noise sources. The results show that although smoothing works surprisingly well, the more sophisticated DNN based estimator shows promising improvements and shortens the performance gap between online and offline processing."}],"citation":{"bibtex":"@inproceedings{Heymann_Drude_Haeb-Umbach_Kinoshita_Nakatani_2018, title={Frame-Online DNN-WPE Dereverberation}, booktitle={IWAENC 2018, Tokio, Japan}, author={Heymann, Jahn and Drude, Lukas and Haeb-Umbach, Reinhold and Kinoshita, Keisuke and Nakatani, Tomohiro}, year={2018} }","ama":"Heymann J, Drude L, Haeb-Umbach R, Kinoshita K, Nakatani T. Frame-Online DNN-WPE Dereverberation. In: <i>IWAENC 2018, Tokio, Japan</i>. ; 2018.","mla":"Heymann, Jahn, et al. “Frame-Online DNN-WPE Dereverberation.” <i>IWAENC 2018, Tokio, Japan</i>, 2018.","short":"J. Heymann, L. Drude, R. Haeb-Umbach, K. Kinoshita, T. Nakatani, in: IWAENC 2018, Tokio, Japan, 2018.","chicago":"Heymann, Jahn, Lukas Drude, Reinhold Haeb-Umbach, Keisuke Kinoshita, and Tomohiro Nakatani. “Frame-Online DNN-WPE Dereverberation.” In <i>IWAENC 2018, Tokio, Japan</i>, 2018.","ieee":"J. Heymann, L. Drude, R. Haeb-Umbach, K. Kinoshita, and T. Nakatani, “Frame-Online DNN-WPE Dereverberation,” in <i>IWAENC 2018, Tokio, Japan</i>, 2018.","apa":"Heymann, J., Drude, L., Haeb-Umbach, R., Kinoshita, K., &#38; Nakatani, T. (2018). Frame-Online DNN-WPE Dereverberation. In <i>IWAENC 2018, Tokio, Japan</i>."},"publication":"IWAENC 2018, Tokio, Japan","oa":"1","department":[{"_id":"54"}],"type":"conference","date_created":"2019-07-12T05:29:10Z"},{"abstract":[{"lang":"eng","text":"We present a block-online multi-channel front end for automatic speech recognition in noisy and reverberated environments. It is an online version of our earlier proposed neural network supported acoustic beamformer, whose coefficients are calculated from noise and speech spatial covariance matrices which are estimated utilizing a neural mask estimator. However, the sparsity of speech in the STFT domain causes problems for the initial beamformer coefficients estimation in some frequency bins due to lack of speech observations. We propose two methods to mitigate this issue. The first is to lower the frequency resolution of the STFT, which comes with the additional advantage of a reduced time window, thus lowering the latency introduced by block processing. The second approach is to smooth beamforming coefficients along the frequency axis, thus exploiting their high interfrequency correlation. With both approaches the gap between offline and block-online beamformer performance, as measured by the word error rate achieved by a downstream speech recognizer, is significantly reduced. Experiments are carried out on two copora, representing noisy (CHiME-4) and noisy reverberant (voiceHome) environments."}],"related_material":{"link":[{"description":"Slides","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Heitkaemper_Slides.pdf","relation":"supplementary_material"}]},"publication":"ITG 2018, Oldenburg, Germany","citation":{"apa":"Heitkaemper, J., Heymann, J., &#38; Haeb-Umbach, R. (2018). Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming. In <i>ITG 2018, Oldenburg, Germany</i>.","mla":"Heitkaemper, Jens, et al. “Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming.” <i>ITG 2018, Oldenburg, Germany</i>, 2018.","ieee":"J. Heitkaemper, J. Heymann, and R. Haeb-Umbach, “Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming,” in <i>ITG 2018, Oldenburg, Germany</i>, 2018.","chicago":"Heitkaemper, Jens, Jahn Heymann, and Reinhold Haeb-Umbach. “Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming.” In <i>ITG 2018, Oldenburg, Germany</i>, 2018.","short":"J. Heitkaemper, J. Heymann, R. Haeb-Umbach, in: ITG 2018, Oldenburg, Germany, 2018.","ama":"Heitkaemper J, Heymann J, Haeb-Umbach R. Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming. In: <i>ITG 2018, Oldenburg, Germany</i>. ; 2018.","bibtex":"@inproceedings{Heitkaemper_Heymann_Haeb-Umbach_2018, title={Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming}, booktitle={ITG 2018, Oldenburg, Germany}, author={Heitkaemper, Jens and Heymann, Jahn and Haeb-Umbach, Reinhold}, year={2018} }"},"type":"conference","department":[{"_id":"54"}],"oa":"1","date_created":"2019-07-12T05:29:13Z","date_updated":"2022-01-06T06:51:11Z","title":"Smoothing along Frequency in Online Neural Network Supported Acoustic Beamforming","status":"public","year":"2018","author":[{"id":"27643","full_name":"Heitkaemper, Jens","last_name":"Heitkaemper","first_name":"Jens"},{"first_name":"Jahn","last_name":"Heymann","full_name":"Heymann, Jahn","id":"9168"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"user_id":"44006","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Heitkaemper_Paper.pdf","open_access":"1"}],"_id":"11837","language":[{"iso":"eng"}]},{"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:29:53Z","abstract":[{"text":"The weighted prediction error (WPE) algorithm has proven to be a very successful dereverberation method for the REVERB challenge. Likewise, neural network based mask estimation for beamforming demonstrated very good noise suppression in the CHiME 3 and CHiME 4 challenges. Recently, it has been shown that this estimator can also be trained to perform dereverberation and denoising jointly. However, up to now a comparison of a neural beamformer and WPE is still missing, so is an investigation into a combination of the two. Therefore, we here provide an extensive evaluation of both and consequently propose variants to integrate deep neural network based beamforming with WPE. For these integrated variants we identify a consistent word error rate (WER) reduction on two distinct databases. In particular, our study shows that deep learning based beamforming benefits from a model-based dereverberation technique (i.e. WPE) and vice versa. Our key findings are: (a) Neural beamforming yields the lower WERs in comparison to WPE the more channels and noise are present. (b) Integration of WPE and a neural beamformer consistently outperforms all stand-alone systems.","lang":"eng"}],"related_material":{"link":[{"description":"Slides","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Drude_Slides.pdf","relation":"supplementary_material"}]},"project":[{"name":"Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"}],"publication":"INTERSPEECH 2018, Hyderabad, India","citation":{"ieee":"L. Drude <i>et al.</i>, “Integration neural network based beamforming and weighted prediction error dereverberation,” in <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","apa":"Drude, L., Boeddeker, C., Heymann, J., Kinoshita, K., Delcroix, M., Nakatani, T., &#38; Haeb-Umbach, R. (2018). Integration neural network based beamforming and weighted prediction error dereverberation. In <i>INTERSPEECH 2018, Hyderabad, India</i>.","short":"L. Drude, C. Boeddeker, J. Heymann, K. Kinoshita, M. Delcroix, T. Nakatani, R. Haeb-Umbach, in: INTERSPEECH 2018, Hyderabad, India, 2018.","chicago":"Drude, Lukas, Christoph Boeddeker, Jahn Heymann, Keisuke Kinoshita, Marc Delcroix, Tomohiro Nakatani, and Reinhold Haeb-Umbach. “Integration Neural Network Based Beamforming and Weighted Prediction Error Dereverberation.” In <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","mla":"Drude, Lukas, et al. “Integration Neural Network Based Beamforming and Weighted Prediction Error Dereverberation.” <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","bibtex":"@inproceedings{Drude_Boeddeker_Heymann_Kinoshita_Delcroix_Nakatani_Haeb-Umbach_2018, title={Integration neural network based beamforming and weighted prediction error dereverberation}, booktitle={INTERSPEECH 2018, Hyderabad, India}, author={Drude, Lukas and Boeddeker, Christoph and Heymann, Jahn and Kinoshita, Keisuke and Delcroix, Marc and Nakatani, Tomohiro and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Drude L, Boeddeker C, Heymann J, et al. Integration neural network based beamforming and weighted prediction error dereverberation. In: <i>INTERSPEECH 2018, Hyderabad, India</i>. ; 2018."},"user_id":"40767","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Drude_Paper.pdf"}],"_id":"11872","language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:11Z","title":"Integration neural network based beamforming and weighted prediction error dereverberation","year":"2018","status":"public","author":[{"first_name":"Lukas","last_name":"Drude","full_name":"Drude, Lukas","id":"11213"},{"full_name":"Boeddeker, Christoph","last_name":"Boeddeker","first_name":"Christoph","id":"40767"},{"full_name":"Heymann, Jahn","first_name":"Jahn","last_name":"Heymann","id":"9168"},{"full_name":"Kinoshita, Keisuke","first_name":"Keisuke","last_name":"Kinoshita"},{"full_name":"Delcroix, Marc","last_name":"Delcroix","first_name":"Marc"},{"full_name":"Nakatani, Tomohiro","last_name":"Nakatani","first_name":"Tomohiro"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"}]},{"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:29:54Z","abstract":[{"lang":"eng","text":"NARA-WPE is a Python software package providing implementations of the weighted prediction error (WPE) dereverberation algorithm. WPE has been shown to be a highly effective tool for speech dereverberation, thus improving the perceptual quality of the signal and improving the recognition performance of downstream automatic speech recognition (ASR). It is suitable both for single-channel and multi-channel applications. The package consist of (1) a Numpy implementation which can easily be integrated into a custom Python toolchain, and (2) a TensorFlow implementation which allows integration into larger computational graphs and enables backpropagation through WPE to train more advanced front-ends. This package comprises of an iterative offline (batch) version, a block-online version, and a frame-online version which can be used in moderately low latency applications, e.g. digital speech assistants."}],"related_material":{"link":[{"description":"Poster","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Drude_Poster.pdf","relation":"supplementary_material"}]},"project":[{"_id":"52","name":"Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"publication":"ITG 2018, Oldenburg, Germany","citation":{"ama":"Drude L, Heymann J, Boeddeker C, Haeb-Umbach R. NARA-WPE: A Python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing. In: <i>ITG 2018, Oldenburg, Germany</i>. ; 2018.","bibtex":"@inproceedings{Drude_Heymann_Boeddeker_Haeb-Umbach_2018, title={NARA-WPE: A Python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing}, booktitle={ITG 2018, Oldenburg, Germany}, author={Drude, Lukas and Heymann, Jahn and Boeddeker, Christoph and Haeb-Umbach, Reinhold}, year={2018} }","mla":"Drude, Lukas, et al. “NARA-WPE: A Python Package for Weighted Prediction Error Dereverberation in Numpy and Tensorflow for Online and Offline Processing.” <i>ITG 2018, Oldenburg, Germany</i>, 2018.","short":"L. Drude, J. Heymann, C. Boeddeker, R. Haeb-Umbach, in: ITG 2018, Oldenburg, Germany, 2018.","chicago":"Drude, Lukas, Jahn Heymann, Christoph Boeddeker, and Reinhold Haeb-Umbach. “NARA-WPE: A Python Package for Weighted Prediction Error Dereverberation in Numpy and Tensorflow for Online and Offline Processing.” In <i>ITG 2018, Oldenburg, Germany</i>, 2018.","apa":"Drude, L., Heymann, J., Boeddeker, C., &#38; Haeb-Umbach, R. (2018). NARA-WPE: A Python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing. In <i>ITG 2018, Oldenburg, Germany</i>.","ieee":"L. Drude, J. Heymann, C. Boeddeker, and R. Haeb-Umbach, “NARA-WPE: A Python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing,” in <i>ITG 2018, Oldenburg, Germany</i>, 2018."},"user_id":"40767","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Drude_Paper.pdf"}],"language":[{"iso":"eng"}],"_id":"11873","date_updated":"2022-01-06T06:51:11Z","title":"NARA-WPE: A Python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing","status":"public","year":"2018","author":[{"first_name":"Lukas","last_name":"Drude","full_name":"Drude, Lukas","id":"11213"},{"id":"9168","first_name":"Jahn","last_name":"Heymann","full_name":"Heymann, Jahn"},{"full_name":"Boeddeker, Christoph","first_name":"Christoph","last_name":"Boeddeker","id":"40767"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}]},{"_id":"11916","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/SpeechCommunication_2018_Walter_Paper.pdf","open_access":"1"}],"user_id":"44006","author":[{"full_name":"Despotovic, Vladimir","first_name":"Vladimir","last_name":"Despotovic"},{"full_name":"Walter, Oliver","first_name":"Oliver","last_name":"Walter"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"}],"year":"2018","title":"Machine learning techniques for semantic analysis of dysarthric speech: An experimental study","status":"public","date_updated":"2022-01-06T06:51:12Z","date_created":"2019-07-12T05:30:44Z","department":[{"_id":"54"}],"oa":"1","type":"journal_article","citation":{"short":"V. Despotovic, O. Walter, R. Haeb-Umbach, Speech Communication 99 (2018) 242-251 (Elsevier B.V.) (2018).","chicago":"Despotovic, Vladimir, Oliver Walter, and Reinhold Haeb-Umbach. “Machine Learning Techniques for Semantic Analysis of Dysarthric Speech: An Experimental Study.” <i>Speech Communication 99 (2018) 242-251 (Elsevier B.V.)</i>, 2018.","apa":"Despotovic, V., Walter, O., &#38; Haeb-Umbach, R. (2018). Machine learning techniques for semantic analysis of dysarthric speech: An experimental study. <i>Speech Communication 99 (2018) 242-251 (Elsevier B.V.)</i>.","ieee":"V. Despotovic, O. Walter, and R. Haeb-Umbach, “Machine learning techniques for semantic analysis of dysarthric speech: An experimental study,” <i>Speech Communication 99 (2018) 242-251 (Elsevier B.V.)</i>, 2018.","ama":"Despotovic V, Walter O, Haeb-Umbach R. Machine learning techniques for semantic analysis of dysarthric speech: An experimental study. <i>Speech Communication 99 (2018) 242-251 (Elsevier BV)</i>. 2018.","bibtex":"@article{Despotovic_Walter_Haeb-Umbach_2018, title={Machine learning techniques for semantic analysis of dysarthric speech: An experimental study}, journal={Speech Communication 99 (2018) 242-251 (Elsevier B.V.)}, author={Despotovic, Vladimir and Walter, Oliver and Haeb-Umbach, Reinhold}, year={2018} }","mla":"Despotovic, Vladimir, et al. “Machine Learning Techniques for Semantic Analysis of Dysarthric Speech: An Experimental Study.” <i>Speech Communication 99 (2018) 242-251 (Elsevier B.V.)</i>, 2018."},"publication":"Speech Communication 99 (2018) 242-251 (Elsevier B.V.)","abstract":[{"lang":"eng","text":"We present an experimental comparison of seven state-of-the-art machine learning algorithms for the task of semantic analysis of spoken input, with a special emphasis on applications for dysarthric speech. Dysarthria is a motor speech disorder, which is characterized by poor articulation of phonemes. In order to cater for these noncanonical phoneme realizations, we employed an unsupervised learning approach to estimate the acoustic models for speech recognition, which does not require a literal transcription of the training data. Even for the subsequent task of semantic analysis, only weak supervision is employed, whereby the training utterance is accompanied by a semantic label only, rather than a literal transcription. Results on two databases, one of them containing dysarthric speech, are presented showing that Markov logic networks and conditional random fields substantially outperform other machine learning approaches. Markov logic networks have proved to be especially robust to recognition errors, which are caused by imprecise articulation in dysarthric speech."}]},{"publication":"ICASSP 2018, Calgary, Canada","citation":{"apa":"Drude, L., von Neumann, T., &#38; Haeb-Umbach, R. (2018). Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation. In <i>ICASSP 2018, Calgary, Canada</i>.","ieee":"L. Drude, T. von Neumann, and R. Haeb-Umbach, “Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation,” in <i>ICASSP 2018, Calgary, Canada</i>, 2018.","chicago":"Drude, Lukas, Thilo von Neumann, and Reinhold Haeb-Umbach. “Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation.” In <i>ICASSP 2018, Calgary, Canada</i>, 2018.","short":"L. Drude, T. von Neumann, R. Haeb-Umbach, in: ICASSP 2018, Calgary, Canada, 2018.","mla":"Drude, Lukas, et al. “Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation.” <i>ICASSP 2018, Calgary, Canada</i>, 2018.","ama":"Drude L, von Neumann T, Haeb-Umbach R. Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation. In: <i>ICASSP 2018, Calgary, Canada</i>. ; 2018.","bibtex":"@inproceedings{Drude_von Neumann_Haeb-Umbach_2018, title={Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation}, booktitle={ICASSP 2018, Calgary, Canada}, author={Drude, Lukas and von Neumann, Thilo and Haeb-Umbach, Reinhold}, year={2018} }"},"related_material":{"link":[{"description":"Slides","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Drude2_Slides.pdf"}]},"abstract":[{"text":"Deep clustering (DC) and deep attractor networks (DANs) are a data-driven way to monaural blind source separation. Both approaches provide astonishing single channel performance but have not yet been generalized to block-online processing. When separating speech in a continuous stream with a block-online algorithm, it needs to be determined in each block which of the output streams belongs to whom. In this contribution we solve this block permutation problem by introducing an additional speaker identification embedding to the DAN model structure. We motivate this model decision by analyzing the embedding topology of DC and DANs and show, that DC and DANs themselves are not sufficient for speaker identification. This model structure (a) improves the signal to distortion ratio (SDR) over a DAN baseline and (b) provides up to 61% and up to 34% relative reduction in permutation error rate and re-identification error rate compared to an i-vector baseline, respectively.","lang":"eng"}],"date_created":"2019-07-30T14:22:53Z","type":"conference","oa":"1","department":[{"_id":"54"}],"title":"Deep Attractor Networks for Speaker Re-Identifikation and Blind Source Separation","status":"public","year":"2018","author":[{"full_name":"Drude, Lukas","first_name":"Lukas","last_name":"Drude","id":"11213"},{"full_name":"von Neumann, Thilo","last_name":"von Neumann","first_name":"Thilo"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"date_updated":"2022-01-06T06:51:24Z","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Drude2_Paper.pdf","open_access":"1"}],"language":[{"iso":"eng"}],"_id":"12898","user_id":"44006"},{"citation":{"mla":"Drude, Lukas, et al. “Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation.” <i>ICASSP 2018, Calgary, Canada</i>, 2018.","ama":"Drude L, Higuchi,  Takuya , Kinoshita K, Nakatani T, Haeb-Umbach R. Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation. In: <i>ICASSP 2018, Calgary, Canada</i>. ; 2018.","bibtex":"@inproceedings{Drude_Higuchi,_Kinoshita_Nakatani_Haeb-Umbach_2018, title={Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation}, booktitle={ICASSP 2018, Calgary, Canada}, author={Drude, Lukas and Higuchi,  Takuya  and Kinoshita, Keisuke  and Nakatani, Tomohiro  and Haeb-Umbach, Reinhold}, year={2018} }","apa":"Drude, L., Higuchi,  Takuya , Kinoshita, K., Nakatani, T., &#38; Haeb-Umbach, R. (2018). Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation. In <i>ICASSP 2018, Calgary, Canada</i>.","ieee":"L. Drude,  Takuya  Higuchi, K. Kinoshita, T. Nakatani, and R. Haeb-Umbach, “Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation,” in <i>ICASSP 2018, Calgary, Canada</i>, 2018.","chicago":"Drude, Lukas,  Takuya  Higuchi, Keisuke  Kinoshita, Tomohiro  Nakatani, and Reinhold Haeb-Umbach. “Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation.” In <i>ICASSP 2018, Calgary, Canada</i>, 2018.","short":"L. Drude,  Takuya  Higuchi, K. Kinoshita, T. Nakatani, R. Haeb-Umbach, in: ICASSP 2018, Calgary, Canada, 2018."},"publication":"ICASSP 2018, Calgary, Canada","related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Drude_Poster.pdf","relation":"supplementary_material","description":"Poster"}]},"abstract":[{"lang":"eng","text":"Deep attractor networks (DANs) are a recently introduced method to blindly separate sources from spectral features of a monaural recording using bidirectional long short-term memory networks (BLSTMs). Due to the nature of BLSTMs, this is inherently not online-ready and resorting to operating on blocks yields a block permutation problem in that the index of each speaker may change between blocks. We here propose the joint modeling of spatial and spectral features to solve the block permutation problem and generalize DANs to multi-channel meeting recordings: The DAN acts as a spectral feature extractor for a subsequent model-based clustering approach. We first analyze different joint models in batch-processing scenarios and finally propose a block-online blind source separation algorithm. The efficacy of the proposed models is demonstrated on reverberant mixtures corrupted by real recordings of multi-channel background noise. We demonstrate that both the proposed batch-processing and the proposed block-online system outperform (a) a spatial-only model with a state-of-the-art frequency permutation solver and (b) a spectral-only model with an oracle block permutation solver in terms of signal to distortion ratio (SDR) gains."}],"date_created":"2019-07-30T14:42:15Z","oa":"1","department":[{"_id":"54"}],"type":"conference","author":[{"full_name":"Drude, Lukas","last_name":"Drude","first_name":"Lukas","id":"11213"},{"full_name":"Higuchi,,  Takuya ","first_name":" Takuya ","last_name":"Higuchi,"},{"full_name":"Kinoshita, Keisuke ","last_name":"Kinoshita","first_name":"Keisuke "},{"full_name":"Nakatani, Tomohiro ","first_name":"Tomohiro ","last_name":"Nakatani"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"title":"Dual Frequency- and Block-Permutation Alignment for Deep Learning Based Block-Online Blind Source Separation","status":"public","year":"2018","date_updated":"2022-01-06T06:51:24Z","language":[{"iso":"eng"}],"_id":"12900","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Drude_Paper.pdf","open_access":"1"}],"user_id":"44006"},{"publication":"ICASSP 2018, Calgary, Canada","citation":{"bibtex":"@inproceedings{Boeddeker_Erdogan_Yoshioka_Haeb-Umbach_2018, title={Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition}, booktitle={ICASSP 2018, Calgary, Canada}, author={Boeddeker, Christoph and Erdogan, Hakan and Yoshioka, Takuya and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Boeddeker C, Erdogan H, Yoshioka T, Haeb-Umbach R. Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition. In: <i>ICASSP 2018, Calgary, Canada</i>. ; 2018.","mla":"Boeddeker, Christoph, et al. “Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition.” <i>ICASSP 2018, Calgary, Canada</i>, 2018.","chicago":"Boeddeker, Christoph, Hakan Erdogan, Takuya Yoshioka, and Reinhold Haeb-Umbach. “Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition.” In <i>ICASSP 2018, Calgary, Canada</i>, 2018.","short":"C. Boeddeker, H. Erdogan, T. Yoshioka, R. Haeb-Umbach, in: ICASSP 2018, Calgary, Canada, 2018.","ieee":"C. Boeddeker, H. Erdogan, T. Yoshioka, and R. Haeb-Umbach, “Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition,” in <i>ICASSP 2018, Calgary, Canada</i>, 2018.","apa":"Boeddeker, C., Erdogan, H., Yoshioka, T., &#38; Haeb-Umbach, R. (2018). Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition. In <i>ICASSP 2018, Calgary, Canada</i>."},"related_material":{"link":[{"description":"Poster","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Boeddeker_Slides.pdf"}]},"abstract":[{"lang":"eng","text":"This work examines acoustic beamformers employing neural networks (NNs) for mask prediction as front-end for automatic speech recognition (ASR) systems for practical scenarios like voice-enabled home devices. To test the versatility of the mask predicting network, the system is evaluated with different recording hardware, different microphone array designs, and different acoustic models of the downstream ASR system. Significant gains in recognition accuracy are obtained in all configurations despite the fact that the NN had been trained on mismatched data. Unlike previous work, the NN is trained on a feature level objective, which gives some performance advantage over a mask related criterion. Furthermore, different approaches for realizing online, or adaptive, NN-based beamforming are explored, where the online algorithms still show significant gains compared to the baseline performance."}],"date_created":"2019-07-30T14:53:58Z","type":"conference","oa":"1","department":[{"_id":"54"}],"status":"public","year":"2018","title":"Exploring Practical Aspects of Neural Mask-Based Beamforming for Far-Field Speech Recognition","author":[{"first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph","id":"40767"},{"first_name":"Hakan","last_name":"Erdogan","full_name":"Erdogan, Hakan"},{"full_name":"Yoshioka, Takuya","first_name":"Takuya","last_name":"Yoshioka"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}],"date_updated":"2022-01-06T06:51:24Z","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICASSP_2018_Boeddeker_Paper.pdf"}],"language":[{"iso":"eng"}],"_id":"12901","user_id":"44006"},{"abstract":[{"lang":"eng","text":"This paper introduces a new open source platform for end-toend speech processing named ESPnet. ESPnet mainly focuses on end-to-end automatic speech recognition (ASR), and adopts widely-used dynamic neural network toolkits, Chainer and Py-Torch, as a main deep learning engine. ESPnet also follows the Kaldi ASR toolkit style for data processing, feature extraction/format, and recipes to provide a complete setup for speech recognition and other speech processing experiments. This paper explains a major architecture of this software platform, several important functionalities, which differentiate ESPnet from other open source ASR toolkits, and experimental results with\r\nmajor ASR benchmarks."}],"publication":"INTERSPEECH 2018, Hyderabad, India","type":"conference","department":[{"_id":"54"}],"file":[{"creator":"huesera","date_created":"2022-02-23T08:03:13Z","date_updated":"2022-02-23T08:03:13Z","relation":"main_file","file_size":288907,"access_level":"open_access","file_name":"INTERSPEECH_2018_Heymann_Paper.pdf","content_type":"application/pdf","file_id":"29954"}],"date_created":"2022-02-21T10:34:37Z","date_updated":"2023-01-11T11:23:19Z","title":"ESPnet: End-to-End Speech Processing Toolkit","year":"2018","author":[{"last_name":"Watanabe","first_name":"Shinji","full_name":"Watanabe, Shinji"},{"last_name":"Hori","first_name":"Takaaki","full_name":"Hori, Takaaki"},{"first_name":"Shigeki","last_name":"Karita","full_name":"Karita, Shigeki"},{"full_name":"Hayashi, Tomoki","first_name":"Tomoki","last_name":"Hayashi"},{"last_name":"Nishitoba","first_name":"Jiro","full_name":"Nishitoba, Jiro"},{"first_name":"Yuya","last_name":"Unno","full_name":"Unno, Yuya"},{"full_name":"Enrique Yalta Soplin, Nelson","first_name":"Nelson","last_name":"Enrique Yalta Soplin"},{"id":"9168","first_name":"Jahn","last_name":"Heymann","full_name":"Heymann, Jahn"},{"full_name":"Wiesner, Matthew","first_name":"Matthew","last_name":"Wiesner"},{"full_name":"Chen, Nanxin","last_name":"Chen","first_name":"Nanxin"},{"first_name":"Adithya","last_name":"Renduchintala","full_name":"Renduchintala, Adithya"},{"full_name":"Ochiai, Tsubasa","last_name":"Ochiai","first_name":"Tsubasa"}],"doi":"10.21437/Interspeech.2018-1456","language":[{"iso":"eng"}],"file_date_updated":"2022-02-23T08:03:13Z","citation":{"mla":"Watanabe, Shinji, et al. “ESPnet: End-to-End Speech Processing Toolkit.” <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018, pp. 2207–2211, doi:<a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">10.21437/Interspeech.2018-1456</a>.","apa":"Watanabe, S., Hori, T., Karita, S., Hayashi, T., Nishitoba, J., Unno, Y., Enrique Yalta Soplin, N., Heymann, J., Wiesner, M., Chen, N., Renduchintala, A., &#38; Ochiai, T. (2018). ESPnet: End-to-End Speech Processing Toolkit. <i>INTERSPEECH 2018, Hyderabad, India</i>, 2207–2211. <a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">https://doi.org/10.21437/Interspeech.2018-1456</a>","ieee":"S. Watanabe <i>et al.</i>, “ESPnet: End-to-End Speech Processing Toolkit,” in <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018, pp. 2207–2211, doi: <a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">10.21437/Interspeech.2018-1456</a>.","short":"S. Watanabe, T. Hori, S. Karita, T. Hayashi, J. Nishitoba, Y. Unno, N. Enrique Yalta Soplin, J. Heymann, M. Wiesner, N. Chen, A. Renduchintala, T. Ochiai, in: INTERSPEECH 2018, Hyderabad, India, 2018, pp. 2207–2211.","ama":"Watanabe S, Hori T, Karita S, et al. ESPnet: End-to-End Speech Processing Toolkit. In: <i>INTERSPEECH 2018, Hyderabad, India</i>. ; 2018:2207–2211. doi:<a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">10.21437/Interspeech.2018-1456</a>","chicago":"Watanabe, Shinji, Takaaki Hori, Shigeki Karita, Tomoki Hayashi, Jiro Nishitoba, Yuya Unno, Nelson Enrique Yalta Soplin, et al. “ESPnet: End-to-End Speech Processing Toolkit.” In <i>INTERSPEECH 2018, Hyderabad, India</i>, 2207–2211, 2018. <a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">https://doi.org/10.21437/Interspeech.2018-1456</a>.","bibtex":"@inproceedings{Watanabe_Hori_Karita_Hayashi_Nishitoba_Unno_Enrique Yalta Soplin_Heymann_Wiesner_Chen_et al._2018, title={ESPnet: End-to-End Speech Processing Toolkit}, DOI={<a href=\"https://doi.org/10.21437/Interspeech.2018-1456\">10.21437/Interspeech.2018-1456</a>}, booktitle={INTERSPEECH 2018, Hyderabad, India}, author={Watanabe, Shinji and Hori, Takaaki and Karita, Shigeki and Hayashi, Tomoki and Nishitoba, Jiro and Unno, Yuya and Enrique Yalta Soplin, Nelson and Heymann, Jahn and Wiesner, Matthew and Chen, Nanxin and et al.}, year={2018}, pages={2207–2211} }"},"oa":"1","has_accepted_license":"1","status":"public","user_id":"59789","ddc":["000"],"page":"2207–2211","_id":"29923"},{"related_material":{"link":[{"description":"Poster","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Heitkaemper_Poster.pdf","relation":"supplementary_material"}]},"quality_controlled":"1","abstract":[{"text":"This contribution presents a speech enhancement system for the CHiME-5 Dinner Party Scenario. The front-end employs multi-channel linear time-variant filtering and achieves its gains without the use of a neural network. We present an adaptation of blind source separation techniques to the CHiME-5 database which we call Guided Source Separation (GSS). Using the baseline acoustic and language model, the combination of Weighted Prediction Error based dereverberation, guided source separation, and beamforming reduces the WER by 10:54% (relative) for the single array track and by 21:12% (relative) on the multiple array track.","lang":"eng"}],"project":[{"name":"Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"}],"publication":"Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India","citation":{"ama":"Boeddeker C, Heitkaemper J, Schmalenstroeer J, Drude L, Heymann J, Haeb-Umbach R. Front-End Processing for the CHiME-5 Dinner Party Scenario. In: <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>. ; 2018.","bibtex":"@inproceedings{Boeddeker_Heitkaemper_Schmalenstroeer_Drude_Heymann_Haeb-Umbach_2018, title={Front-End Processing for the CHiME-5 Dinner Party Scenario}, booktitle={Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India}, author={Boeddeker, Christoph and Heitkaemper, Jens and Schmalenstroeer, Joerg and Drude, Lukas and Heymann, Jahn and Haeb-Umbach, Reinhold}, year={2018} }","mla":"Boeddeker, Christoph, et al. “Front-End Processing for the CHiME-5 Dinner Party Scenario.” <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>, 2018.","chicago":"Boeddeker, Christoph, Jens Heitkaemper, Joerg Schmalenstroeer, Lukas Drude, Jahn Heymann, and Reinhold Haeb-Umbach. “Front-End Processing for the CHiME-5 Dinner Party Scenario.” In <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>, 2018.","short":"C. Boeddeker, J. Heitkaemper, J. Schmalenstroeer, L. Drude, J. Heymann, R. Haeb-Umbach, in: Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India, 2018.","apa":"Boeddeker, C., Heitkaemper, J., Schmalenstroeer, J., Drude, L., Heymann, J., &#38; Haeb-Umbach, R. (2018). Front-End Processing for the CHiME-5 Dinner Party Scenario. <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>.","ieee":"C. Boeddeker, J. Heitkaemper, J. Schmalenstroeer, L. Drude, J. Heymann, and R. Haeb-Umbach, “Front-End Processing for the CHiME-5 Dinner Party Scenario,” 2018."},"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-30T14:35:15Z","date_updated":"2023-10-26T08:14:15Z","status":"public","year":"2018","title":"Front-End Processing for the CHiME-5 Dinner Party Scenario","author":[{"last_name":"Boeddeker","first_name":"Christoph","full_name":"Boeddeker, Christoph","id":"40767"},{"full_name":"Heitkaemper, Jens","first_name":"Jens","last_name":"Heitkaemper","id":"27643"},{"id":"460","first_name":"Joerg","last_name":"Schmalenstroeer","full_name":"Schmalenstroeer, Joerg"},{"first_name":"Lukas","last_name":"Drude","full_name":"Drude, Lukas","id":"11213"},{"first_name":"Jahn","last_name":"Heymann","full_name":"Heymann, Jahn"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"user_id":"460","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Heitkaemper_Paper.pdf"}],"_id":"12899","language":[{"iso":"eng"}]},{"date_updated":"2023-10-26T08:15:32Z","author":[{"last_name":"Afifi","first_name":"Haitham","full_name":"Afifi, Haitham","id":"65718"},{"last_name":"Schmalenstroeer","first_name":"Joerg","full_name":"Schmalenstroeer, Joerg","id":"460"},{"id":"16256","first_name":"Joerg","last_name":"Ullmann","full_name":"Ullmann, Joerg"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"},{"id":"126","full_name":"Karl, Holger","first_name":"Holger","last_name":"Karl"}],"status":"public","title":"MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks","year":"2018","user_id":"460","_id":"6859","language":[{"iso":"eng"}],"page":"1-5","project":[{"_id":"27","name":"Akustische Sensornetzwerke - Teilprojekt "},{"_id":"27","name":"Akustische Sensornetzwerke - Teilprojekt \"Verteilte akustische Signalverarbeitung über funkbasierte Sensornetzwerke"}],"abstract":[{"lang":"eng","text":"Signal processing in WASNs is based on a software framework for hosting the algorithms as well as on a set of wireless connected devices representing the hardware. Each of the nodes contributes memory, processing power, communication bandwidth and some sensor information for the tasks to be solved on the network. \r\nIn this paper we present our MARVELO framework for distributed signal processing. It is intended for transforming existing centralized implementations into distributed versions. To this end, the software only needs a block-oriented implementation, which MARVELO picks-up and distributes on the network. Additionally, our sensor node hardware and the audio interfaces responsible for multi-channel recordings are presented."}],"quality_controlled":"1","citation":{"mla":"Afifi, Haitham, et al. “MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks.” <i>Speech Communication; 13th ITG-Symposium</i>, 2018, pp. 1–5.","bibtex":"@inproceedings{Afifi_Schmalenstroeer_Ullmann_Haeb-Umbach_Karl_2018, title={MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks}, booktitle={Speech Communication; 13th ITG-Symposium}, author={Afifi, Haitham and Schmalenstroeer, Joerg and Ullmann, Joerg and Haeb-Umbach, Reinhold and Karl, Holger}, year={2018}, pages={1–5} }","ama":"Afifi H, Schmalenstroeer J, Ullmann J, Haeb-Umbach R, Karl H. MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks. In: <i>Speech Communication; 13th ITG-Symposium</i>. ; 2018:1-5.","ieee":"H. Afifi, J. Schmalenstroeer, J. Ullmann, R. Haeb-Umbach, and H. Karl, “MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks,” in <i>Speech Communication; 13th ITG-Symposium</i>, 2018, pp. 1–5.","apa":"Afifi, H., Schmalenstroeer, J., Ullmann, J., Haeb-Umbach, R., &#38; Karl, H. (2018). MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks. <i>Speech Communication; 13th ITG-Symposium</i>, 1–5.","chicago":"Afifi, Haitham, Joerg Schmalenstroeer, Joerg Ullmann, Reinhold Haeb-Umbach, and Holger Karl. “MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks.” In <i>Speech Communication; 13th ITG-Symposium</i>, 1–5, 2018.","short":"H. Afifi, J. Schmalenstroeer, J. Ullmann, R. Haeb-Umbach, H. Karl, in: Speech Communication; 13th ITG-Symposium, 2018, pp. 1–5."},"publication":"Speech Communication; 13th ITG-Symposium","department":[{"_id":"75"},{"_id":"54"}],"type":"conference","date_created":"2019-01-17T15:47:35Z"},{"date_created":"2019-07-12T05:27:29Z","oa":"1","department":[{"_id":"54"}],"type":"conference","citation":{"ama":"Grimm C, Breddermann T, Farhoud R, Fei T, Warsitz E, Haeb-Umbach R. Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar. In: <i>International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018</i>. ; 2018.","bibtex":"@inproceedings{Grimm_Breddermann_Farhoud_Fei_Warsitz_Haeb-Umbach_2018, title={Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar}, booktitle={International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018}, author={Grimm, Christopher and Breddermann, Tobias and Farhoud, Ridha and Fei, Tai and Warsitz, Ernst and Haeb-Umbach, Reinhold}, year={2018} }","mla":"Grimm, Christopher, et al. “Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar.” <i>International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018</i>, 2018.","short":"C. Grimm, T. Breddermann, R. Farhoud, T. Fei, E. Warsitz, R. Haeb-Umbach, in: International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018, 2018.","chicago":"Grimm, Christopher, Tobias Breddermann, Ridha Farhoud, Tai Fei, Ernst Warsitz, and Reinhold Haeb-Umbach. “Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar.” In <i>International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018</i>, 2018.","apa":"Grimm, C., Breddermann, T., Farhoud, R., Fei, T., Warsitz, E., &#38; Haeb-Umbach, R. (2018). Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar. <i>International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018</i>.","ieee":"C. Grimm, T. Breddermann, R. Farhoud, T. Fei, E. Warsitz, and R. Haeb-Umbach, “Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar,” 2018."},"publication":"International Conference on Microwaves for Intelligent Mobility (ICMIM) 2018","quality_controlled":"1","abstract":[{"lang":"eng","text":"In this paper, we present a neural network based classification algorithm for the discrimination of moving from stationary targets in the sight of an automotive radar sensor. Compared to existing algorithms, the proposed algorithm can take into account multiple local radar targets instead of performing classification inference on each target individually resulting in superior discrimination accuracy, especially suitable for non rigid objects, like pedestrians, which in general have a wide velocity spread when multiple targets are detected."}],"_id":"11747","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/ICMIM_2018_Haeb-Umbach_Paper.pdf","open_access":"1"}],"user_id":"242","author":[{"full_name":"Grimm, Christopher","first_name":"Christopher","last_name":"Grimm"},{"first_name":"Tobias","last_name":"Breddermann","full_name":"Breddermann, Tobias"},{"last_name":"Farhoud","first_name":"Ridha","full_name":"Farhoud, Ridha"},{"full_name":"Fei, Tai","last_name":"Fei","first_name":"Tai"},{"first_name":"Ernst","last_name":"Warsitz","full_name":"Warsitz, Ernst"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"year":"2018","status":"public","title":"Discrimination of Stationary from Moving Targets with Recurrent Neural Networks in Automotive Radar","date_updated":"2023-11-20T16:37:39Z"},{"date_updated":"2023-11-22T08:29:22Z","title":"Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery","status":"public","year":"2018","author":[{"id":"14169","last_name":"Glarner","first_name":"Thomas","full_name":"Glarner, Thomas"},{"full_name":"Hanebrink, Patrick","last_name":"Hanebrink","first_name":"Patrick"},{"id":"34851","full_name":"Ebbers, Janek","last_name":"Ebbers","first_name":"Janek"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"user_id":"34851","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Glarner_Paper.pdf","open_access":"1"}],"_id":"11907","language":[{"iso":"eng"}],"abstract":[{"text":"The invention of the Variational Autoencoder enables the application of Neural Networks to a wide range of tasks in unsupervised learning, including the field of Acoustic Unit Discovery (AUD). The recently proposed Hidden Markov Model Variational Autoencoder (HMMVAE) allows a joint training of a neural network based feature extractor and a structured prior for the latent space given by a Hidden Markov Model. It has been shown that the HMMVAE significantly outperforms pure GMM-HMM based systems on the AUD task. However, the HMMVAE cannot autonomously infer the number of acoustic units and thus relies on the GMM-HMM system for initialization. This paper introduces the Bayesian Hidden Markov Model Variational Autoencoder (BHMMVAE) which solves these issues by embedding the HMMVAE in a Bayesian framework with a Dirichlet Process Prior for the distribution of the acoustic units, and diagonal or full-covariance Gaussians as emission distributions. Experiments on TIMIT and Xitsonga show that the BHMMVAE is able to autonomously infer a reasonable number of acoustic units, can be initialized without supervision by a GMM-HMM system, achieves computationally efficient stochastic variational inference by using natural gradient descent, and, additionally, improves the AUD performance over the HMMVAE.","lang":"eng"}],"related_material":{"link":[{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Glarner_Slides.pdf","description":"Slides"}]},"quality_controlled":"1","publication":"INTERSPEECH 2018, Hyderabad, India","citation":{"chicago":"Glarner, Thomas, Patrick Hanebrink, Janek Ebbers, and Reinhold Haeb-Umbach. “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” In <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","short":"T. Glarner, P. Hanebrink, J. Ebbers, R. Haeb-Umbach, in: INTERSPEECH 2018, Hyderabad, India, 2018.","ieee":"T. Glarner, P. Hanebrink, J. Ebbers, and R. Haeb-Umbach, “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery,” 2018.","apa":"Glarner, T., Hanebrink, P., Ebbers, J., &#38; Haeb-Umbach, R. (2018). Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. <i>INTERSPEECH 2018, Hyderabad, India</i>.","bibtex":"@inproceedings{Glarner_Hanebrink_Ebbers_Haeb-Umbach_2018, title={Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery}, booktitle={INTERSPEECH 2018, Hyderabad, India}, author={Glarner, Thomas and Hanebrink, Patrick and Ebbers, Janek and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Glarner T, Hanebrink P, Ebbers J, Haeb-Umbach R. Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. In: <i>INTERSPEECH 2018, Hyderabad, India</i>. ; 2018.","mla":"Glarner, Thomas, et al. “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018."},"type":"conference","department":[{"_id":"54"}],"oa":"1","date_created":"2019-07-12T05:30:34Z"},{"year":"2018","status":"public","title":"Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach","author":[{"first_name":"Joerg","last_name":"Schmalenstroeer","full_name":"Schmalenstroeer, Joerg","id":"460"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"date_updated":"2023-10-26T08:12:33Z","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/Eusipco_2018_Schmalenstroeer_Paper.pdf"}],"language":[{"iso":"eng"}],"_id":"11838","user_id":"460","publication":"26th European Signal Processing Conference (EUSIPCO 2018)","citation":{"mla":"Schmalenstroeer, Joerg, and Reinhold Haeb-Umbach. “Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach.” <i>26th European Signal Processing Conference (EUSIPCO 2018)</i>, 2018.","bibtex":"@inproceedings{Schmalenstroeer_Haeb-Umbach_2018, title={Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach}, booktitle={26th European Signal Processing Conference (EUSIPCO 2018)}, author={Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Schmalenstroeer J, Haeb-Umbach R. Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach. In: <i>26th European Signal Processing Conference (EUSIPCO 2018)</i>. ; 2018.","ieee":"J. Schmalenstroeer and R. Haeb-Umbach, “Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach,” 2018.","apa":"Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2018). Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach. <i>26th European Signal Processing Conference (EUSIPCO 2018)</i>.","short":"J. Schmalenstroeer, R. Haeb-Umbach, in: 26th European Signal Processing Conference (EUSIPCO 2018), 2018.","chicago":"Schmalenstroeer, Joerg, and Reinhold Haeb-Umbach. “Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach.” In <i>26th European Signal Processing Conference (EUSIPCO 2018)</i>, 2018."},"quality_controlled":"1","abstract":[{"lang":"eng","text":"Distributed sensor data acquisition usually encompasses data sampling by the individual devices, where each of them has its own oscillator driving the local sampling process, resulting in slightly different sampling rates at the individual sensor nodes. Nevertheless, for certain downstream signal processing tasks it is important to compensate even for small sampling rate offsets. Aligning the sampling rates of oscillators which differ only by a few parts-per-million, is, however, challenging and quite different from traditional multirate signal processing tasks. In this paper we propose to transfer a precise but computationally demanding time domain approach, inspired by the Nyquist-Shannon sampling theorem, to an efficient frequency domain implementation. To this end a buffer control is employed which compensates for sampling offsets which are multiples of the sampling period, while a digital filter, realized by the wellknown Overlap-Save method, handles the fractional part of the sampling phase offset. With experiments on artificially misaligned data we investigate the parametrization, the efficiency, and the induced distortions of the proposed resampling method. It is shown that a favorable compromise between residual distortion and computational complexity is achieved, compared to other sampling rate offset compensation techniques."}],"date_created":"2019-07-12T05:29:14Z","type":"conference","oa":"1","department":[{"_id":"54"}]},{"citation":{"mla":"Kitza, Markus, et al. “The RWTH/UPB System Combination for the CHiME 2018 Workshop.” <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>, 2018.","ama":"Kitza M, Michel W, Boeddeker C, et al. The RWTH/UPB System Combination for the CHiME 2018 Workshop. In: <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>. ; 2018.","bibtex":"@inproceedings{Kitza_Michel_Boeddeker_Heitkaemper_Menne_Schlüter_Ney_Schmalenstroeer_Drude_Heymann_et al._2018, title={The RWTH/UPB System Combination for the CHiME 2018 Workshop}, booktitle={Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India}, author={Kitza, Markus and Michel, Wilfried and Boeddeker, Christoph and Heitkaemper, Jens and Menne, Tobias and Schlüter, Ralf and Ney, Hermann and Schmalenstroeer, Joerg and Drude, Lukas and Heymann, Jahn and et al.}, year={2018} }","apa":"Kitza, M., Michel, W., Boeddeker, C., Heitkaemper, J., Menne, T., Schlüter, R., Ney, H., Schmalenstroeer, J., Drude, L., Heymann, J., &#38; Haeb-Umbach, R. (2018). The RWTH/UPB System Combination for the CHiME 2018 Workshop. <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>.","ieee":"M. Kitza <i>et al.</i>, “The RWTH/UPB System Combination for the CHiME 2018 Workshop,” 2018.","short":"M. Kitza, W. Michel, C. Boeddeker, J. Heitkaemper, T. Menne, R. Schlüter, H. Ney, J. Schmalenstroeer, L. Drude, J. Heymann, R. Haeb-Umbach, in: Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India, 2018.","chicago":"Kitza, Markus, Wilfried Michel, Christoph Boeddeker, Jens Heitkaemper, Tobias Menne, Ralf Schlüter, Hermann Ney, et al. “The RWTH/UPB System Combination for the CHiME 2018 Workshop.” In <i>Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India</i>, 2018."},"publication":"Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India","abstract":[{"text":"This paper describes the systems for the single-array track and the multiple-array track of the 5th CHiME Challenge. The final system is a combination of multiple systems, using Confusion Network Combination (CNC). The different systems presented here are utilizing different front-ends and training sets for a Bidirectional Long Short-Term Memory (BLSTM) Acoustic Model (AM). The front-end was replaced by enhancements provided by Paderborn University [1]. The back-end has been implemented using RASR [2] and RETURNN [3]. Additionally, a system combination including the hypothesis word graphs from the system of the submission [1] has been performed, which results in the final best system.","lang":"eng"}],"quality_controlled":"1","date_created":"2019-07-12T05:29:58Z","department":[{"_id":"54"}],"oa":"1","type":"conference","author":[{"full_name":"Kitza, Markus","last_name":"Kitza","first_name":"Markus"},{"full_name":"Michel, Wilfried","last_name":"Michel","first_name":"Wilfried"},{"id":"40767","last_name":"Boeddeker","first_name":"Christoph","full_name":"Boeddeker, Christoph"},{"last_name":"Heitkaemper","first_name":"Jens","full_name":"Heitkaemper, Jens","id":"27643"},{"full_name":"Menne, Tobias","first_name":"Tobias","last_name":"Menne"},{"full_name":"Schlüter, Ralf","last_name":"Schlüter","first_name":"Ralf"},{"first_name":"Hermann","last_name":"Ney","full_name":"Ney, Hermann"},{"id":"460","full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer"},{"first_name":"Lukas","last_name":"Drude","full_name":"Drude, Lukas","id":"11213"},{"id":"9168","last_name":"Heymann","first_name":"Jahn","full_name":"Heymann, Jahn"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"year":"2018","status":"public","title":"The RWTH/UPB System Combination for the CHiME 2018 Workshop","date_updated":"2023-10-26T08:12:14Z","language":[{"iso":"eng"}],"_id":"11876","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Heitkaemper_RWTH_Paper.pdf","open_access":"1"}],"user_id":"460"},{"date_created":"2019-07-12T05:29:11Z","department":[{"_id":"54"}],"oa":"1","type":"conference","citation":{"ieee":"J. Ebbers, J. Heitkaemper, J. Schmalenstroeer, and R. Haeb-Umbach, “Benchmarking Neural Network Architectures for Acoustic Sensor Networks,” 2018.","apa":"Ebbers, J., Heitkaemper, J., Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2018). Benchmarking Neural Network Architectures for Acoustic Sensor Networks. <i>ITG 2018, Oldenburg, Germany</i>.","chicago":"Ebbers, Janek, Jens Heitkaemper, Joerg Schmalenstroeer, and Reinhold Haeb-Umbach. “Benchmarking Neural Network Architectures for Acoustic Sensor Networks.” In <i>ITG 2018, Oldenburg, Germany</i>, 2018.","short":"J. Ebbers, J. Heitkaemper, J. Schmalenstroeer, R. Haeb-Umbach, in: ITG 2018, Oldenburg, Germany, 2018.","mla":"Ebbers, Janek, et al. “Benchmarking Neural Network Architectures for Acoustic Sensor Networks.” <i>ITG 2018, Oldenburg, Germany</i>, 2018.","bibtex":"@inproceedings{Ebbers_Heitkaemper_Schmalenstroeer_Haeb-Umbach_2018, title={Benchmarking Neural Network Architectures for Acoustic Sensor Networks}, booktitle={ITG 2018, Oldenburg, Germany}, author={Ebbers, Janek and Heitkaemper, Jens and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Ebbers J, Heitkaemper J, Schmalenstroeer J, Haeb-Umbach R. Benchmarking Neural Network Architectures for Acoustic Sensor Networks. In: <i>ITG 2018, Oldenburg, Germany</i>. ; 2018."},"publication":"ITG 2018, Oldenburg, Germany","related_material":{"link":[{"description":"Poster","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Ebbers_Poster.pdf"}]},"quality_controlled":"1","abstract":[{"lang":"eng","text":"Due to their distributed nature wireless acoustic sensor networks offer great potential for improved signal acquisition, processing and classification for applications such as monitoring and surveillance, home automation, or hands-free telecommunication. To reduce the communication demand with a central server and to raise the privacy level it is desirable to perform processing at node level. The limited processing and memory capabilities on a sensor node, however, stand in contrast to the compute and memory intensive deep learning algorithms used in modern speech and audio processing. In this work, we perform benchmarking of commonly used convolutional and recurrent neural network architectures on a Raspberry Pi based acoustic sensor node. We show that it is possible to run medium-sized neural network topologies used for speech enhancement and speech recognition in real time. For acoustic event recognition, where predictions in a lower temporal resolution are sufficient, it is even possible to run current state-of-the-art deep convolutional models with a real-time-factor of 0:11."}],"language":[{"iso":"eng"}],"_id":"11836","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Ebbers_Paper.pdf"}],"user_id":"460","author":[{"id":"34851","full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers"},{"id":"27643","last_name":"Heitkaemper","first_name":"Jens","full_name":"Heitkaemper, Jens"},{"full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer","id":"460"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"status":"public","title":"Benchmarking Neural Network Architectures for Acoustic Sensor Networks","year":"2018","date_updated":"2023-10-26T08:12:40Z"}]
