[{"date_updated":"2023-11-22T08:30:12Z","year":"2019","title":"Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision","author":[{"first_name":"Janek","last_name":"Ebbers","full_name":"Ebbers, Janek","id":"34851"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}],"language":[{"iso":"eng"}],"abstract":[{"text":"In this paper we present our audio tagging system for the DCASE 2019 Challenge Task 2. We propose a model consisting of a convolutional front end using log-mel-energies as input features, a recurrent neural network sequence encoder and a fully connected classifier network outputting an activity probability for each of the 80 considered event classes. Due to the recurrent neural network, which encodes a whole sequence into a single vector, our model is able to process sequences of varying lengths. The model is trained with only little manually labeled training data and a larger amount of automatically labeled web data, which hence suffers from label noise. To efficiently train the model with the provided data we use various data augmentation to prevent overfitting and improve generalization. Our best submitted system achieves a label-weighted label-ranking average precision (lwlrap) of 75.5% on the private test set which is an absolute improvement of 21.7% over the baseline. This system scored the second place in the teams ranking of the DCASE 2019 Challenge Task 2 and the fifth place in the Kaggle competition “Freesound Audio Tagging 2019” with more than 400 participants. After the challenge ended we further improved performance to 76.5% lwlrap setting a new state-of-the-art on this dataset.","lang":"eng"}],"publication":"DCASE2019 Workshop, New York, USA","type":"conference","department":[{"_id":"54"}],"file":[{"date_created":"2020-02-05T10:18:06Z","creator":"huesera","content_type":"application/pdf","file_id":"15795","file_size":184967,"access_level":"open_access","file_name":"DCASE_2019_WS_Ebbers_Paper.pdf","date_updated":"2020-02-05T10:18:06Z","relation":"main_file"}],"date_created":"2020-02-05T10:16:03Z","has_accepted_license":"1","status":"public","user_id":"34851","ddc":["000"],"_id":"15794","quality_controlled":"1","project":[{"_id":"52","name":"Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"file_date_updated":"2020-02-05T10:18:06Z","citation":{"apa":"Ebbers, J., &#38; Haeb-Umbach, R. (2019). Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision. <i>DCASE2019 Workshop, New York, USA</i>.","ieee":"J. Ebbers and R. Haeb-Umbach, “Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision,” 2019.","short":"J. Ebbers, R. Haeb-Umbach, in: DCASE2019 Workshop, New York, USA, 2019.","chicago":"Ebbers, Janek, and Reinhold Haeb-Umbach. “Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision.” In <i>DCASE2019 Workshop, New York, USA</i>, 2019.","mla":"Ebbers, Janek, and Reinhold Haeb-Umbach. “Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision.” <i>DCASE2019 Workshop, New York, USA</i>, 2019.","ama":"Ebbers J, Haeb-Umbach R. Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision. In: <i>DCASE2019 Workshop, New York, USA</i>. ; 2019.","bibtex":"@inproceedings{Ebbers_Haeb-Umbach_2019, title={Convolutional Recurrent Neural Network and Data Augmentation for Audio Tagging with Noisy Labels and Minimal Supervision}, booktitle={DCASE2019 Workshop, New York, USA}, author={Ebbers, Janek and Haeb-Umbach, Reinhold}, year={2019} }"},"oa":"1"},{"department":[{"_id":"54"}],"type":"conference","date_created":"2020-02-05T10:20:17Z","file":[{"creator":"huesera","date_created":"2020-02-05T10:21:39Z","relation":"main_file","date_updated":"2020-02-05T10:21:39Z","file_name":"CAMSAP_2019_WS_Ebbers_Paper.pdf","access_level":"open_access","file_size":311887,"file_id":"15797","content_type":"application/pdf"}],"abstract":[{"text":"In this paper we consider human daily activity recognition using an acoustic sensor network (ASN) which consists of nodes distributed in a home environment. Assuming that the ASN is permanently recording, the vast majority of recordings is silence. Therefore, we propose to employ a computationally efficient two-stage sound recognition system, consisting of an initial sound activity detection (SAD) and a subsequent sound event classification (SEC), which is only activated once sound activity has been detected. We show how a low-latency activity detector with high temporal resolution can be trained from weak labels with low temporal resolution. We further demonstrate the advantage of using spatial features for the subsequent event classification task.","lang":"eng"}],"publication":"CAMSAP 2019, Guadeloupe, West Indies","language":[{"iso":"eng"}],"date_updated":"2023-11-22T08:29:58Z","author":[{"id":"34851","full_name":"Ebbers, Janek","last_name":"Ebbers","first_name":"Janek"},{"id":"11213","full_name":"Drude, Lukas","last_name":"Drude","first_name":"Lukas"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"},{"full_name":"Brendel, Andreas","first_name":"Andreas","last_name":"Brendel"},{"full_name":"Kellermann, Walter","first_name":"Walter","last_name":"Kellermann"}],"year":"2019","title":"Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks","oa":"1","project":[{"name":"Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"}],"quality_controlled":"1","citation":{"mla":"Ebbers, Janek, et al. “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks.” <i>CAMSAP 2019, Guadeloupe, West Indies</i>, 2019.","ama":"Ebbers J, Drude L, Haeb-Umbach R, Brendel A, Kellermann W. Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks. In: <i>CAMSAP 2019, Guadeloupe, West Indies</i>. ; 2019.","bibtex":"@inproceedings{Ebbers_Drude_Haeb-Umbach_Brendel_Kellermann_2019, title={Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks}, booktitle={CAMSAP 2019, Guadeloupe, West Indies}, author={Ebbers, Janek and Drude, Lukas and Haeb-Umbach, Reinhold and Brendel, Andreas and Kellermann, Walter}, year={2019} }","apa":"Ebbers, J., Drude, L., Haeb-Umbach, R., Brendel, A., &#38; Kellermann, W. (2019). Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks. <i>CAMSAP 2019, Guadeloupe, West Indies</i>.","ieee":"J. Ebbers, L. Drude, R. Haeb-Umbach, A. Brendel, and W. Kellermann, “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks,” 2019.","chicago":"Ebbers, Janek, Lukas Drude, Reinhold Haeb-Umbach, Andreas Brendel, and Walter Kellermann. “Weakly Supervised Sound Activity Detection and Event Classification in Acoustic Sensor Networks.” In <i>CAMSAP 2019, Guadeloupe, West Indies</i>, 2019.","short":"J. Ebbers, L. Drude, R. Haeb-Umbach, A. Brendel, W. Kellermann, in: CAMSAP 2019, Guadeloupe, West Indies, 2019."},"file_date_updated":"2020-02-05T10:21:39Z","ddc":["000"],"user_id":"34851","_id":"15796","has_accepted_license":"1","status":"public"},{"ddc":["000"],"user_id":"34851","_id":"15792","has_accepted_license":"1","status":"public","oa":"1","quality_controlled":"1","file_date_updated":"2020-02-05T10:11:40Z","citation":{"mla":"Nelus, Alexandru, et al. “Privacy-Preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification.” <i>INTERSPEECH 2019, Graz, Austria</i>, 2019.","bibtex":"@inproceedings{Nelus_Ebbers_Haeb-Umbach_Martin_2019, title={Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification}, booktitle={INTERSPEECH 2019, Graz, Austria}, author={Nelus, Alexandru and Ebbers, Janek and Haeb-Umbach, Reinhold and Martin, Rainer}, year={2019} }","ama":"Nelus A, Ebbers J, Haeb-Umbach R, Martin R. Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification. In: <i>INTERSPEECH 2019, Graz, Austria</i>. ; 2019.","ieee":"A. Nelus, J. Ebbers, R. Haeb-Umbach, and R. Martin, “Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification,” 2019.","apa":"Nelus, A., Ebbers, J., Haeb-Umbach, R., &#38; Martin, R. (2019). Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification. <i>INTERSPEECH 2019, Graz, Austria</i>.","short":"A. Nelus, J. Ebbers, R. Haeb-Umbach, R. Martin, in: INTERSPEECH 2019, Graz, Austria, 2019.","chicago":"Nelus, Alexandru, Janek Ebbers, Reinhold Haeb-Umbach, and Rainer Martin. “Privacy-Preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification.” In <i>INTERSPEECH 2019, Graz, Austria</i>, 2019."},"language":[{"iso":"eng"}],"date_updated":"2023-11-22T08:27:55Z","title":"Privacy-preserving Variational Information Feature Extraction for Domestic Activity Monitoring Versus Speaker Identification","year":"2019","author":[{"full_name":"Nelus, Alexandru","last_name":"Nelus","first_name":"Alexandru"},{"first_name":"Janek","last_name":"Ebbers","full_name":"Ebbers, Janek","id":"34851"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"},{"last_name":"Martin","first_name":"Rainer","full_name":"Martin, Rainer"}],"type":"conference","department":[{"_id":"54"}],"file":[{"date_created":"2020-02-05T10:11:40Z","creator":"huesera","content_type":"application/pdf","file_id":"15793","date_updated":"2020-02-05T10:11:40Z","relation":"main_file","file_size":454600,"access_level":"open_access","file_name":"INTERSPEECH_2019_Ebbers_Paper.pdf"}],"date_created":"2020-02-05T10:07:53Z","abstract":[{"text":"In this paper we highlight the privacy risks entailed in deep neural network feature extraction for domestic activity monitoring. We employ the baseline system proposed in the Task 5 of the DCASE 2018 challenge and simulate a feature interception attack by an eavesdropper who wants to perform speaker identification. We then propose to reduce the aforementioned privacy risks by introducing a variational information feature extraction scheme that allows for good activity monitoring performance while at the same time minimizing the information of the feature representation, thus restricting speaker identification attempts. We analyze the resulting model’s composite loss function and the budget scaling factor used to control the balance between the performance of the trusted and attacker tasks. It is empirically demonstrated that the proposed method reduces speaker identification privacy risks without significantly deprecating the performance of domestic activity monitoring tasks.","lang":"eng"}],"publication":"INTERSPEECH 2019, Graz, Austria"},{"user_id":"44006","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/Daga_2018_Ebbers_Paper.pdf"}],"language":[{"iso":"eng"}],"_id":"11760","date_updated":"2022-01-06T06:51:08Z","title":"Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection","year":"2018","status":"public","author":[{"full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers","id":"34851"},{"first_name":"Alexandru","last_name":"Nelus","full_name":"Nelus, Alexandru"},{"full_name":"Martin, Rainer","first_name":"Rainer","last_name":"Martin"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:27:43Z","abstract":[{"text":"Acoustic event detection, i.e., the task of assigning a human interpretable label to a segment of audio, has only recently attracted increased interest in the research community. Driven by the DCASE challenges and the availability of large-scale audio datasets, the state-of-the-art has progressed rapidly with deep-learning-based classi- fiers dominating the field. Because several potential use cases favor a realization on distributed sensor nodes, e.g. ambient assisted living applications, habitat monitoring or surveillance, we are concerned with two issues here. Firstly the classification performance of such systems and secondly the computing resources required to achieve a certain performance considering node level feature extraction. In this contribution we look at the balance between the two criteria by employing traditional techniques and different deep learning architectures, including convolutional and recurrent models in the context of real life everyday audio recordings in realistic, however challenging, multisource conditions.","lang":"eng"}],"publication":"DAGA 2018, München","citation":{"apa":"Ebbers, J., Nelus, A., Martin, R., &#38; Haeb-Umbach, R. (2018). Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection. In <i>DAGA 2018, München</i>.","ieee":"J. Ebbers, A. Nelus, R. Martin, and R. Haeb-Umbach, “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection,” in <i>DAGA 2018, München</i>, 2018.","chicago":"Ebbers, Janek, Alexandru Nelus, Rainer Martin, and Reinhold Haeb-Umbach. “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection.” In <i>DAGA 2018, München</i>, 2018.","short":"J. Ebbers, A. Nelus, R. Martin, R. Haeb-Umbach, in: DAGA 2018, München, 2018.","mla":"Ebbers, Janek, et al. “Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection.” <i>DAGA 2018, München</i>, 2018.","ama":"Ebbers J, Nelus A, Martin R, Haeb-Umbach R. Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection. In: <i>DAGA 2018, München</i>. ; 2018.","bibtex":"@inproceedings{Ebbers_Nelus_Martin_Haeb-Umbach_2018, title={Evaluation of Modulation-MFCC Features and DNN Classification for Acoustic Event Detection}, booktitle={DAGA 2018, München}, author={Ebbers, Janek and Nelus, Alexandru and Martin, Rainer and Haeb-Umbach, Reinhold}, year={2018} }"}},{"citation":{"ieee":"T. Glarner, P. Hanebrink, J. Ebbers, and R. Haeb-Umbach, “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery,” 2018.","apa":"Glarner, T., Hanebrink, P., Ebbers, J., &#38; Haeb-Umbach, R. (2018). Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. <i>INTERSPEECH 2018, Hyderabad, India</i>.","chicago":"Glarner, Thomas, Patrick Hanebrink, Janek Ebbers, and Reinhold Haeb-Umbach. “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” In <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","short":"T. Glarner, P. Hanebrink, J. Ebbers, R. Haeb-Umbach, in: INTERSPEECH 2018, Hyderabad, India, 2018.","mla":"Glarner, Thomas, et al. “Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” <i>INTERSPEECH 2018, Hyderabad, India</i>, 2018.","bibtex":"@inproceedings{Glarner_Hanebrink_Ebbers_Haeb-Umbach_2018, title={Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery}, booktitle={INTERSPEECH 2018, Hyderabad, India}, author={Glarner, Thomas and Hanebrink, Patrick and Ebbers, Janek and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Glarner T, Hanebrink P, Ebbers J, Haeb-Umbach R. Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. In: <i>INTERSPEECH 2018, Hyderabad, India</i>. ; 2018."},"publication":"INTERSPEECH 2018, Hyderabad, India","related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Glarner_Slides.pdf","relation":"supplementary_material","description":"Slides"}]},"quality_controlled":"1","abstract":[{"text":"The invention of the Variational Autoencoder enables the application of Neural Networks to a wide range of tasks in unsupervised learning, including the field of Acoustic Unit Discovery (AUD). The recently proposed Hidden Markov Model Variational Autoencoder (HMMVAE) allows a joint training of a neural network based feature extractor and a structured prior for the latent space given by a Hidden Markov Model. It has been shown that the HMMVAE significantly outperforms pure GMM-HMM based systems on the AUD task. However, the HMMVAE cannot autonomously infer the number of acoustic units and thus relies on the GMM-HMM system for initialization. This paper introduces the Bayesian Hidden Markov Model Variational Autoencoder (BHMMVAE) which solves these issues by embedding the HMMVAE in a Bayesian framework with a Dirichlet Process Prior for the distribution of the acoustic units, and diagonal or full-covariance Gaussians as emission distributions. Experiments on TIMIT and Xitsonga show that the BHMMVAE is able to autonomously infer a reasonable number of acoustic units, can be initialized without supervision by a GMM-HMM system, achieves computationally efficient stochastic variational inference by using natural gradient descent, and, additionally, improves the AUD performance over the HMMVAE.","lang":"eng"}],"date_created":"2019-07-12T05:30:34Z","department":[{"_id":"54"}],"oa":"1","type":"conference","author":[{"first_name":"Thomas","last_name":"Glarner","full_name":"Glarner, Thomas","id":"14169"},{"first_name":"Patrick","last_name":"Hanebrink","full_name":"Hanebrink, Patrick"},{"id":"34851","full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"}],"title":"Full Bayesian Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery","year":"2018","status":"public","date_updated":"2023-11-22T08:29:22Z","language":[{"iso":"eng"}],"_id":"11907","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/INTERSPEECH_2018_Glarner_Paper.pdf"}],"user_id":"34851"},{"_id":"11836","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Ebbers_Paper.pdf"}],"user_id":"460","author":[{"id":"34851","full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers"},{"first_name":"Jens","last_name":"Heitkaemper","full_name":"Heitkaemper, Jens","id":"27643"},{"id":"460","last_name":"Schmalenstroeer","first_name":"Joerg","full_name":"Schmalenstroeer, Joerg"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"title":"Benchmarking Neural Network Architectures for Acoustic Sensor Networks","status":"public","year":"2018","date_updated":"2023-10-26T08:12:40Z","date_created":"2019-07-12T05:29:11Z","department":[{"_id":"54"}],"oa":"1","type":"conference","citation":{"mla":"Ebbers, Janek, et al. “Benchmarking Neural Network Architectures for Acoustic Sensor Networks.” <i>ITG 2018, Oldenburg, Germany</i>, 2018.","bibtex":"@inproceedings{Ebbers_Heitkaemper_Schmalenstroeer_Haeb-Umbach_2018, title={Benchmarking Neural Network Architectures for Acoustic Sensor Networks}, booktitle={ITG 2018, Oldenburg, Germany}, author={Ebbers, Janek and Heitkaemper, Jens and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2018} }","ama":"Ebbers J, Heitkaemper J, Schmalenstroeer J, Haeb-Umbach R. Benchmarking Neural Network Architectures for Acoustic Sensor Networks. In: <i>ITG 2018, Oldenburg, Germany</i>. ; 2018.","ieee":"J. Ebbers, J. Heitkaemper, J. Schmalenstroeer, and R. Haeb-Umbach, “Benchmarking Neural Network Architectures for Acoustic Sensor Networks,” 2018.","apa":"Ebbers, J., Heitkaemper, J., Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2018). Benchmarking Neural Network Architectures for Acoustic Sensor Networks. <i>ITG 2018, Oldenburg, Germany</i>.","chicago":"Ebbers, Janek, Jens Heitkaemper, Joerg Schmalenstroeer, and Reinhold Haeb-Umbach. “Benchmarking Neural Network Architectures for Acoustic Sensor Networks.” In <i>ITG 2018, Oldenburg, Germany</i>, 2018.","short":"J. Ebbers, J. Heitkaemper, J. Schmalenstroeer, R. Haeb-Umbach, in: ITG 2018, Oldenburg, Germany, 2018."},"publication":"ITG 2018, Oldenburg, Germany","related_material":{"link":[{"description":"Poster","url":"https://groups.uni-paderborn.de/nt/pubs/2018/ITG_2018_Ebbers_Poster.pdf","relation":"supplementary_material"}]},"quality_controlled":"1","abstract":[{"text":"Due to their distributed nature wireless acoustic sensor networks offer great potential for improved signal acquisition, processing and classification for applications such as monitoring and surveillance, home automation, or hands-free telecommunication. To reduce the communication demand with a central server and to raise the privacy level it is desirable to perform processing at node level. The limited processing and memory capabilities on a sensor node, however, stand in contrast to the compute and memory intensive deep learning algorithms used in modern speech and audio processing. In this work, we perform benchmarking of commonly used convolutional and recurrent neural network architectures on a Raspberry Pi based acoustic sensor node. We show that it is possible to run medium-sized neural network topologies used for speech enhancement and speech recognition in real time. For acoustic event recognition, where predictions in a lower temporal resolution are sufficient, it is even possible to run current state-of-the-art deep convolutional models with a real-time-factor of 0:11.","lang":"eng"}]},{"publication":"INTERSPEECH 2017, Stockholm, Schweden","citation":{"mla":"Ebbers, Janek, et al. “Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” <i>INTERSPEECH 2017, Stockholm, Schweden</i>, 2017.","apa":"Ebbers, J., Heymann, J., Drude, L., Glarner, T., Haeb-Umbach, R., &#38; Raj, B. (2017). Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. <i>INTERSPEECH 2017, Stockholm, Schweden</i>.","ieee":"J. Ebbers, J. Heymann, L. Drude, T. Glarner, R. Haeb-Umbach, and B. Raj, “Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery,” 2017.","short":"J. Ebbers, J. Heymann, L. Drude, T. Glarner, R. Haeb-Umbach, B. Raj, in: INTERSPEECH 2017, Stockholm, Schweden, 2017.","ama":"Ebbers J, Heymann J, Drude L, Glarner T, Haeb-Umbach R, Raj B. Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery. In: <i>INTERSPEECH 2017, Stockholm, Schweden</i>. ; 2017.","chicago":"Ebbers, Janek, Jahn Heymann, Lukas Drude, Thomas Glarner, Reinhold Haeb-Umbach, and Bhiksha Raj. “Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery.” In <i>INTERSPEECH 2017, Stockholm, Schweden</i>, 2017.","bibtex":"@inproceedings{Ebbers_Heymann_Drude_Glarner_Haeb-Umbach_Raj_2017, title={Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery}, booktitle={INTERSPEECH 2017, Stockholm, Schweden}, author={Ebbers, Janek and Heymann, Jahn and Drude, Lukas and Glarner, Thomas and Haeb-Umbach, Reinhold and Raj, Bhiksha}, year={2017} }"},"related_material":{"link":[{"description":"Poster","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2017/INTERSPEECH_2017_Ebbers_poster.pdf"},{"description":"Slides","relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2017/INTERSPEECH_2017_Ebbers_slides.pdf"}]},"abstract":[{"text":"Variational Autoencoders (VAEs) have been shown to provide efficient neural-network-based approximate Bayesian inference for observation models for which exact inference is intractable. Its extension, the so-called Structured VAE (SVAE) allows inference in the presence of both discrete and continuous latent variables. Inspired by this extension, we developed a VAE with Hidden Markov Models (HMMs) as latent models. We applied the resulting HMM-VAE to the task of acoustic unit discovery in a zero resource scenario. Starting from an initial model based on variational inference in an HMM with Gaussian Mixture Model (GMM) emission probabilities, the accuracy of the acoustic unit discovery could be significantly improved by the HMM-VAE. In doing so we were able to demonstrate for an unsupervised learning task what is well-known in the supervised learning case: Neural networks provide superior modeling power compared to GMMs.","lang":"eng"}],"quality_controlled":"1","date_created":"2019-07-12T05:27:42Z","type":"conference","oa":"1","department":[{"_id":"54"}],"status":"public","year":"2017","title":"Hidden Markov Model Variational Autoencoder for Acoustic Unit Discovery","author":[{"id":"34851","last_name":"Ebbers","first_name":"Janek","full_name":"Ebbers, Janek"},{"full_name":"Heymann, Jahn","first_name":"Jahn","last_name":"Heymann","id":"9168"},{"id":"11213","last_name":"Drude","first_name":"Lukas","full_name":"Drude, Lukas"},{"id":"14169","full_name":"Glarner, Thomas","first_name":"Thomas","last_name":"Glarner"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"},{"last_name":"Raj","first_name":"Bhiksha","full_name":"Raj, Bhiksha"}],"date_updated":"2023-11-22T08:29:06Z","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2017/INTERSPEECH_2017_Ebbers_paper.pdf","open_access":"1"}],"_id":"11759","language":[{"iso":"eng"}],"user_id":"34851"}]
