[{"_id":"11810","language":[{"iso":"eng"}],"user_id":"44006","title":"BLSTM supported GEV Beamformer Front-End for the 3RD CHiME Challenge","status":"public","year":"2015","author":[{"id":"9168","full_name":"Heymann, Jahn","last_name":"Heymann","first_name":"Jahn"},{"id":"11213","first_name":"Lukas","last_name":"Drude","full_name":"Drude, Lukas"},{"full_name":"Chinaev, Aleksej","first_name":"Aleksej","last_name":"Chinaev"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"date_updated":"2022-01-06T06:51:09Z","date_created":"2019-07-12T05:28:41Z","type":"conference","department":[{"_id":"54"}],"publication":"Automatic Speech Recognition and Understanding Workshop (ASRU 2015)","citation":{"apa":"Heymann, J., Drude, L., Chinaev, A., &#38; Haeb-Umbach, R. (2015). BLSTM supported GEV Beamformer Front-End for the 3RD CHiME Challenge. In <i>Automatic Speech Recognition and Understanding Workshop (ASRU 2015)</i>.","ieee":"J. Heymann, L. Drude, A. Chinaev, and R. Haeb-Umbach, “BLSTM supported GEV Beamformer Front-End for the 3RD CHiME Challenge,” in <i>Automatic Speech Recognition and Understanding Workshop (ASRU 2015)</i>, 2015.","short":"J. Heymann, L. Drude, A. Chinaev, R. Haeb-Umbach, in: Automatic Speech Recognition and Understanding Workshop (ASRU 2015), 2015.","chicago":"Heymann, Jahn, Lukas Drude, Aleksej Chinaev, and Reinhold Haeb-Umbach. “BLSTM Supported GEV Beamformer Front-End for the 3RD CHiME Challenge.” In <i>Automatic Speech Recognition and Understanding Workshop (ASRU 2015)</i>, 2015.","mla":"Heymann, Jahn, et al. “BLSTM Supported GEV Beamformer Front-End for the 3RD CHiME Challenge.” <i>Automatic Speech Recognition and Understanding Workshop (ASRU 2015)</i>, 2015.","ama":"Heymann J, Drude L, Chinaev A, Haeb-Umbach R. BLSTM supported GEV Beamformer Front-End for the 3RD CHiME Challenge. In: <i>Automatic Speech Recognition and Understanding Workshop (ASRU 2015)</i>. ; 2015.","bibtex":"@inproceedings{Heymann_Drude_Chinaev_Haeb-Umbach_2015, title={BLSTM supported GEV Beamformer Front-End for the 3RD CHiME Challenge}, booktitle={Automatic Speech Recognition and Understanding Workshop (ASRU 2015)}, author={Heymann, Jahn and Drude, Lukas and Chinaev, Aleksej and Haeb-Umbach, Reinhold}, year={2015} }"}},{"date_created":"2019-07-12T05:28:45Z","keyword":["codecs","signal denoising","speech recognition","Bayesian feature enhancement","denoising autoencoder","reverberant ASR","single-channel speech recognition","speaker to microphone distances","unsupervised adaptation","Adaptation models","Noise reduction","Reverberation","Speech","Speech recognition","Training","deep neuronal networks","denoising autoencoder","feature enhancement","robust speech recognition"],"type":"conference","oa":"1","department":[{"_id":"54"}],"publication":"Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference on","citation":{"mla":"Heymann, Jahn, et al. “Unsupervised Adaptation of a Denoising Autoencoder by Bayesian Feature Enhancement for Reverberant Asr under Mismatch Conditions.” <i>Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference On</i>, 2015, pp. 5053–57, doi:<a href=\"https://doi.org/10.1109/ICASSP.2015.7178933\">10.1109/ICASSP.2015.7178933</a>.","bibtex":"@inproceedings{Heymann_Haeb-Umbach_Golik_Schlueter_2015, title={Unsupervised adaptation of a denoising autoencoder by Bayesian Feature Enhancement for reverberant asr under mismatch conditions}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2015.7178933\">10.1109/ICASSP.2015.7178933</a>}, booktitle={Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference on}, author={Heymann, Jahn and Haeb-Umbach, Reinhold and Golik, P. and Schlueter, R.}, year={2015}, pages={5053–5057} }","ama":"Heymann J, Haeb-Umbach R, Golik P, Schlueter R. Unsupervised adaptation of a denoising autoencoder by Bayesian Feature Enhancement for reverberant asr under mismatch conditions. In: <i>Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference On</i>. ; 2015:5053-5057. doi:<a href=\"https://doi.org/10.1109/ICASSP.2015.7178933\">10.1109/ICASSP.2015.7178933</a>","ieee":"J. Heymann, R. Haeb-Umbach, P. Golik, and R. Schlueter, “Unsupervised adaptation of a denoising autoencoder by Bayesian Feature Enhancement for reverberant asr under mismatch conditions,” in <i>Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference on</i>, 2015, pp. 5053–5057.","apa":"Heymann, J., Haeb-Umbach, R., Golik, P., &#38; Schlueter, R. (2015). Unsupervised adaptation of a denoising autoencoder by Bayesian Feature Enhancement for reverberant asr under mismatch conditions. In <i>Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference on</i> (pp. 5053–5057). <a href=\"https://doi.org/10.1109/ICASSP.2015.7178933\">https://doi.org/10.1109/ICASSP.2015.7178933</a>","chicago":"Heymann, Jahn, Reinhold Haeb-Umbach, P. Golik, and R. Schlueter. “Unsupervised Adaptation of a Denoising Autoencoder by Bayesian Feature Enhancement for Reverberant Asr under Mismatch Conditions.” In <i>Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference On</i>, 5053–57, 2015. <a href=\"https://doi.org/10.1109/ICASSP.2015.7178933\">https://doi.org/10.1109/ICASSP.2015.7178933</a>.","short":"J. Heymann, R. Haeb-Umbach, P. Golik, R. Schlueter, in: Acoustics, Speech and Signal Processing (ICASSP), 2015 IEEE International Conference On, 2015, pp. 5053–5057."},"abstract":[{"lang":"eng","text":"The parametric Bayesian Feature Enhancement (BFE) and a datadriven Denoising Autoencoder (DA) both bring performance gains in severe single-channel speech recognition conditions. The first can be adjusted to different conditions by an appropriate parameter setting, while the latter needs to be trained on conditions similar to the ones expected at decoding time, making it vulnerable to a mismatch between training and test conditions. We use a DNN backend and study reverberant ASR under three types of mismatch conditions: different room reverberation times, different speaker to microphone distances and the difference between artificially reverberated data and the recordings in a reverberant environment. We show that for these mismatch conditions BFE can provide the targets for a DA. This unsupervised adaptation provides a performance gain over the direct use of BFE and even enables to compensate for the mismatch of real and simulated reverberant data."}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2015/hey_icassp_2015.pdf","open_access":"1"}],"page":"5053-5057","_id":"11813","language":[{"iso":"eng"}],"doi":"10.1109/ICASSP.2015.7178933","user_id":"44006","status":"public","title":"Unsupervised adaptation of a denoising autoencoder by Bayesian Feature Enhancement for reverberant asr under mismatch conditions","year":"2015","author":[{"id":"9168","full_name":"Heymann, Jahn","first_name":"Jahn","last_name":"Heymann"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"},{"full_name":"Golik, P.","last_name":"Golik","first_name":"P."},{"first_name":"R.","last_name":"Schlueter","full_name":"Schlueter, R."}],"date_updated":"2022-01-06T06:51:09Z"},{"date_updated":"2022-01-06T06:51:11Z","author":[{"full_name":"Jacob, Florian","last_name":"Jacob","first_name":"Florian"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"title":"Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network","year":"2015","status":"public","user_id":"44006","_id":"11830","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2015/JaHa2015.pdf"}],"abstract":[{"text":"Joint audio-visual speaker tracking requires that the locations of microphones and cameras are known and that they are given in a common coordinate system. Sensor self-localization algorithms, however, are usually separately developed for either the acoustic or the visual modality and return their positions in a modality specific coordinate system, often with an unknown rotation, scaling and translation between the two. In this paper we propose two techniques to determine the positions of acoustic sensors in a common coordinate system, based on audio-visual correlates, i.e., events that are localized by both, microphones and cameras separately. The first approach maps the output of an acoustic self-calibration algorithm by estimating rotation, scale and translation to the visual coordinate system, while the second solves a joint system of equations with acoustic and visual directions of arrival as input. The evaluation of the two strategies reveals that joint calibration outperforms the mapping approach and achieves an overall calibration error of 0.20m even in reverberant environments.","lang":"eng"}],"citation":{"short":"F. Jacob, R. Haeb-Umbach, ArXiv E-Prints (2015).","chicago":"Jacob, Florian, and Reinhold Haeb-Umbach. “Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network.” <i>ArXiv E-Prints</i>, 2015.","apa":"Jacob, F., &#38; Haeb-Umbach, R. (2015). Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network. <i>ArXiv E-Prints</i>.","ieee":"F. Jacob and R. Haeb-Umbach, “Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network,” <i>ArXiv e-prints</i>, 2015.","ama":"Jacob F, Haeb-Umbach R. Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network. <i>ArXiv e-prints</i>. 2015.","bibtex":"@article{Jacob_Haeb-Umbach_2015, title={Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network}, journal={ArXiv e-prints}, author={Jacob, Florian and Haeb-Umbach, Reinhold}, year={2015} }","mla":"Jacob, Florian, and Reinhold Haeb-Umbach. “Absolute Geometry Calibration of Distributed Microphone Arrays in an Audio-Visual Sensor Network.” <i>ArXiv E-Prints</i>, 2015."},"publication":"ArXiv e-prints","department":[{"_id":"54"}],"oa":"1","type":"journal_article","date_created":"2019-07-12T05:29:05Z"},{"date_created":"2019-07-12T05:29:49Z","type":"book","department":[{"_id":"54"}],"oa":"1","citation":{"chicago":"Li, Jinyu, Li Deng, Reinhold Haeb-Umbach, and Y. Gong. <i>Robust Automatic Speech Recognition</i>. Elsevier, 2015.","short":"J. Li, L. Deng, R. Haeb-Umbach, Y. Gong, Robust Automatic Speech Recognition, Elsevier, 2015.","ieee":"J. Li, L. Deng, R. Haeb-Umbach, and Y. Gong, <i>Robust Automatic Speech Recognition</i>. Elsevier, 2015.","apa":"Li, J., Deng, L., Haeb-Umbach, R., &#38; Gong, Y. (2015). <i>Robust Automatic Speech Recognition</i>. Elsevier.","bibtex":"@book{Li_Deng_Haeb-Umbach_Gong_2015, title={Robust Automatic Speech Recognition}, publisher={Elsevier}, author={Li, Jinyu and Deng, Li and Haeb-Umbach, Reinhold and Gong, Y.}, year={2015} }","ama":"Li J, Deng L, Haeb-Umbach R, Gong Y. <i>Robust Automatic Speech Recognition</i>. Elsevier; 2015.","mla":"Li, Jinyu, et al. <i>Robust Automatic Speech Recognition</i>. Elsevier, 2015."},"related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2015/RASR_Chap5.pdf","relation":"supplementary_material","description":"Sample-Chapter"},{"description":"Store","url":"http://store.elsevier.com/9780128023983","relation":"supplementary_material"}]},"main_file_link":[{"url":"http://store.elsevier.com/Robust-Automatic-Speech-Recognition/Jinyu-Li/isbn-9780128023983/","open_access":"1"}],"publisher":"Elsevier","_id":"11868","language":[{"iso":"eng"}],"user_id":"44006","status":"public","year":"2015","title":"Robust Automatic Speech Recognition","author":[{"full_name":"Li, Jinyu","last_name":"Li","first_name":"Jinyu"},{"last_name":"Deng","first_name":"Li","full_name":"Deng, Li"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"},{"first_name":"Y.","last_name":"Gong","full_name":"Gong, Y."}],"date_updated":"2022-01-06T06:51:11Z"},{"author":[{"full_name":"Marchi, Erik","first_name":"Erik","last_name":"Marchi"},{"full_name":"Schuller, Bjoern","first_name":"Bjoern","last_name":"Schuller"},{"last_name":"Baron-Cohen","first_name":"Simon","full_name":"Baron-Cohen, Simon"},{"first_name":"Ofer","last_name":"Golan","full_name":"Golan, Ofer"},{"last_name":"Boelte","first_name":"Sven","full_name":"Boelte, Sven"},{"last_name":"Arora","first_name":"Prerna","full_name":"Arora, Prerna"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"year":"2015","status":"public","title":"Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages","date_updated":"2022-01-06T06:51:11Z","language":[{"iso":"eng"}],"_id":"11875","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2015/MaScBaOfSvPrHa.pdf","open_access":"1"}],"user_id":"44006","citation":{"ama":"Marchi E, Schuller B, Baron-Cohen S, et al. Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages. In: <i>INTERSPEECH 2015</i>. ; 2015.","bibtex":"@inproceedings{Marchi_Schuller_Baron-Cohen_Golan_Boelte_Arora_Haeb-Umbach_2015, title={Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages}, booktitle={INTERSPEECH 2015}, author={Marchi, Erik and Schuller, Bjoern and Baron-Cohen, Simon and Golan, Ofer and Boelte, Sven and Arora, Prerna and Haeb-Umbach, Reinhold}, year={2015} }","mla":"Marchi, Erik, et al. “Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages.” <i>INTERSPEECH 2015</i>, 2015.","chicago":"Marchi, Erik, Bjoern Schuller, Simon Baron-Cohen, Ofer Golan, Sven Boelte, Prerna Arora, and Reinhold Haeb-Umbach. “Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages.” In <i>INTERSPEECH 2015</i>, 2015.","short":"E. Marchi, B. Schuller, S. Baron-Cohen, O. Golan, S. Boelte, P. Arora, R. Haeb-Umbach, in: INTERSPEECH 2015, 2015.","apa":"Marchi, E., Schuller, B., Baron-Cohen, S., Golan, O., Boelte, S., Arora, P., &#38; Haeb-Umbach, R. (2015). Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages. In <i>INTERSPEECH 2015</i>.","ieee":"E. Marchi <i>et al.</i>, “Typicality and Emotion in the Voice of Children with Autism Spectrum Condition: Evidence Across Three Languages,” in <i>INTERSPEECH 2015</i>, 2015."},"publication":"INTERSPEECH 2015","abstract":[{"text":"Only a few studies exist on automatic emotion analysis of speech from children with Autism Spectrum Conditions (ASC). Out of these, some preliminary studies have recently focused on comparing the relevance of selected prosodic features against large sets of acoustic, spectral, and cepstral features; however, no study so far provided a comparison of performances across different languages. The present contribution aims to fill this white spot in the literature and provide insight by extensive evaluations carried out on three databases of prompted phrases collected in English, Swedish, and Hebrew, inducing nine emotion categories embedded in short-stories. The datasets contain speech of children with ASC and typically developing children under the same conditions. We evaluate automatic diagnosis and recognition of emotions in atypical childrens voice over the nine categories including binary valence/arousal discrimination.","lang":"eng"}],"date_created":"2019-07-12T05:29:57Z","department":[{"_id":"54"}],"oa":"1","type":"conference"},{"related_material":{"link":[{"description":"Poster","url":"https://groups.uni-paderborn.de/nt/pubs/2015/WaDrHa15_Poster.pdf","relation":"supplementary_material"}]},"abstract":[{"text":"In this paper we present a source counting algorithm to determine the number of speakers in a speech mixture. In our proposed method, we model the histogram of estimated directions of arrival with a nonparametric Bayesian infinite Gaussian mixture model. As an alternative to classical model selection criteria and to avoid specifying the maximum number of mixture components in advance, a Dirichlet process prior is employed over the mixture components. This allows to automatically determine the optimal number of mixture components that most probably model the observations. We demonstrate by experiments that this model outperforms a parametric approach using a finite Gaussian mixture model with a Dirichlet distribution prior over the mixture weights.","lang":"eng"}],"publication":"40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)","citation":{"short":"O. Walter, L. Drude, R. Haeb-Umbach, in: 40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015), 2015.","chicago":"Walter, Oliver, Lukas Drude, and Reinhold Haeb-Umbach. “Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an Infinite Gaussian Mixture Model.” In <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>, 2015.","apa":"Walter, O., Drude, L., &#38; Haeb-Umbach, R. (2015). Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an infinite Gaussian Mixture Model. In <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>.","ieee":"O. Walter, L. Drude, and R. Haeb-Umbach, “Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an infinite Gaussian Mixture Model,” in <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>, 2015.","ama":"Walter O, Drude L, Haeb-Umbach R. Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an infinite Gaussian Mixture Model. In: <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>. ; 2015.","bibtex":"@inproceedings{Walter_Drude_Haeb-Umbach_2015, title={Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an infinite Gaussian Mixture Model}, booktitle={40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)}, author={Walter, Oliver and Drude, Lukas and Haeb-Umbach, Reinhold}, year={2015} }","mla":"Walter, Oliver, et al. “Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an Infinite Gaussian Mixture Model.” <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>, 2015."},"type":"conference","department":[{"_id":"54"}],"oa":"1","date_created":"2019-07-12T05:30:47Z","date_updated":"2022-01-06T06:51:12Z","title":"Source Counting in Speech Mixtures by Nonparametric Bayesian Estimation of an infinite Gaussian Mixture Model","status":"public","year":"2015","author":[{"full_name":"Walter, Oliver","first_name":"Oliver","last_name":"Walter"},{"full_name":"Drude, Lukas","first_name":"Lukas","last_name":"Drude","id":"11213"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"user_id":"44006","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2015/WaDrHa15.pdf"}],"language":[{"iso":"eng"}],"_id":"11919"},{"user_id":"44006","doi":"http://dx.doi.org/10.1007/s13218-015-0372-1","_id":"11922","language":[{"iso":"eng"}],"page":"1-13","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2015/WaHaMoPaHa15.pdf"}],"date_updated":"2022-01-06T06:51:12Z","author":[{"full_name":"Walter, Oliver","first_name":"Oliver","last_name":"Walter"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"},{"last_name":"Mokbel","first_name":"Bassam","full_name":"Mokbel, Bassam"},{"full_name":"Paassen, Benjamin","first_name":"Benjamin","last_name":"Paassen"},{"full_name":"Hammer, Barbara","last_name":"Hammer","first_name":"Barbara"}],"status":"public","title":"Autonomous Learning of Representations","year":"2015","department":[{"_id":"54"}],"oa":"1","type":"journal_article","keyword":["Representation learning","Metric learning","Deep representation","Spoken language"],"date_created":"2019-07-12T05:30:51Z","abstract":[{"text":"Besides the core learning algorithm itself, one major question in machine learning is how to best encode given training data such that the learning technology can efficiently learn based thereon and generalize to novel data. While classical approaches often rely on a hand coded data representation, the topic of autonomous representation or feature learning plays a major role in modern learning architectures. The goal of this contribution is to give an overview about different principles of autonomous feature learning, and to exemplify two principles based on two recent examples: autonomous metric learning for sequences, and autonomous learning of a deep representation for spoken language, respectively.","lang":"eng"}],"citation":{"ieee":"O. Walter, R. Haeb-Umbach, B. Mokbel, B. Paassen, and B. Hammer, “Autonomous Learning of Representations,” <i>KI - Kuenstliche Intelligenz</i>, pp. 1–13, 2015.","apa":"Walter, O., Haeb-Umbach, R., Mokbel, B., Paassen, B., &#38; Hammer, B. (2015). Autonomous Learning of Representations. <i>KI - Kuenstliche Intelligenz</i>, 1–13. <a href=\"http://dx.doi.org/10.1007/s13218-015-0372-1\">http://dx.doi.org/10.1007/s13218-015-0372-1</a>","mla":"Walter, Oliver, et al. “Autonomous Learning of Representations.” <i>KI - Kuenstliche Intelligenz</i>, 2015, pp. 1–13, doi:<a href=\"http://dx.doi.org/10.1007/s13218-015-0372-1\">http://dx.doi.org/10.1007/s13218-015-0372-1</a>.","bibtex":"@article{Walter_Haeb-Umbach_Mokbel_Paassen_Hammer_2015, title={Autonomous Learning of Representations}, DOI={<a href=\"http://dx.doi.org/10.1007/s13218-015-0372-1\">http://dx.doi.org/10.1007/s13218-015-0372-1</a>}, journal={KI - Kuenstliche Intelligenz}, author={Walter, Oliver and Haeb-Umbach, Reinhold and Mokbel, Bassam and Paassen, Benjamin and Hammer, Barbara}, year={2015}, pages={1–13} }","chicago":"Walter, Oliver, Reinhold Haeb-Umbach, Bassam Mokbel, Benjamin Paassen, and Barbara Hammer. “Autonomous Learning of Representations.” <i>KI - Kuenstliche Intelligenz</i>, 2015, 1–13. <a href=\"http://dx.doi.org/10.1007/s13218-015-0372-1\">http://dx.doi.org/10.1007/s13218-015-0372-1</a>.","ama":"Walter O, Haeb-Umbach R, Mokbel B, Paassen B, Hammer B. Autonomous Learning of Representations. <i>KI - Kuenstliche Intelligenz</i>. 2015:1-13. doi:<a href=\"http://dx.doi.org/10.1007/s13218-015-0372-1\">http://dx.doi.org/10.1007/s13218-015-0372-1</a>","short":"O. Walter, R. Haeb-Umbach, B. Mokbel, B. Paassen, B. Hammer, KI - Kuenstliche Intelligenz (2015) 1–13."},"publication":"KI - Kuenstliche Intelligenz"},{"user_id":"44006","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2015/WaHaStHi.pdf"}],"language":[{"iso":"eng"}],"_id":"11923","date_updated":"2022-01-06T06:51:12Z","year":"2015","title":"Lexicon Discovery for Language Preservation using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)","status":"public","author":[{"last_name":"Walter","first_name":"Oliver","full_name":"Walter, Oliver"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"},{"first_name":"Jan","last_name":"Strunk","full_name":"Strunk, Jan"},{"full_name":"P. Himmelmann, Nikolaus ","first_name":"Nikolaus ","last_name":"P. Himmelmann"}],"type":"report","department":[{"_id":"54"}],"oa":"1","date_created":"2019-07-12T05:30:52Z","abstract":[{"text":"In this paper we show that recently developed algorithms for unsupervised word segmentation can be a valuable tool for the documentation of endangered languages. We applied an unsupervised word segmentation algorithm based on a nested Pitman-Yor language model to two austronesian languages, Wooi and Waima'a. The algorithm was then modified and parameterized to cater the needs of linguists for high precision of lexical discovery: We obtained a lexicon precision of of 69.2\\% and 67.5\\% for Wooi and Waima'a, respectively, if single-letter words and words found less than three times were discarded. A comparison with an English word segmentation task showed comparable performance, verifying that the assumptions underlying the Pitman-Yor language model, the universality of Zipf's law and the power of n-gram structures, do also hold for languages as exotic as Wooi and Waima'a.","lang":"eng"}],"citation":{"chicago":"Walter, Oliver, Reinhold Haeb-Umbach, Jan Strunk, and Nikolaus  P. Himmelmann. <i>Lexicon Discovery for Language Preservation Using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)</i>, 2015.","short":"O. Walter, R. Haeb-Umbach, J. Strunk, N. P. Himmelmann, Lexicon Discovery for Language Preservation Using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01), 2015.","ieee":"O. Walter, R. Haeb-Umbach, J. Strunk, and N. P. Himmelmann, <i>Lexicon Discovery for Language Preservation using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)</i>. 2015.","apa":"Walter, O., Haeb-Umbach, R., Strunk, J., &#38; P. Himmelmann, N. (2015). <i>Lexicon Discovery for Language Preservation using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)</i>.","bibtex":"@book{Walter_Haeb-Umbach_Strunk_P. Himmelmann_2015, title={Lexicon Discovery for Language Preservation using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)}, author={Walter, Oliver and Haeb-Umbach, Reinhold and Strunk, Jan and P. Himmelmann, Nikolaus }, year={2015} }","ama":"Walter O, Haeb-Umbach R, Strunk J, P. Himmelmann N. <i>Lexicon Discovery for Language Preservation Using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)</i>.; 2015.","mla":"Walter, Oliver, et al. <i>Lexicon Discovery for Language Preservation Using Unsupervised Word Segmentation with Pitman-Yor Language Models (FGNT-2015-01)</i>. 2015."}},{"department":[{"_id":"54"}],"oa":"1","type":"conference","date_created":"2019-07-12T05:29:55Z","quality_controlled":"1","citation":{"bibtex":"@inproceedings{Hoang_Schmalenstroeer_Haeb-Umbach_2015, title={Aligning training models with smartphone properties in WiFi fingerprinting based indoor localization}, booktitle={40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)}, author={Hoang, Manh Kha and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2015} }","ama":"Hoang MK, Schmalenstroeer J, Haeb-Umbach R. Aligning training models with smartphone properties in WiFi fingerprinting based indoor localization. In: <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>. ; 2015.","mla":"Hoang, Manh Kha, et al. “Aligning Training Models with Smartphone Properties in WiFi Fingerprinting Based Indoor Localization.” <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>, 2015.","chicago":"Hoang, Manh Kha, Joerg Schmalenstroeer, and Reinhold Haeb-Umbach. “Aligning Training Models with Smartphone Properties in WiFi Fingerprinting Based Indoor Localization.” In <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>, 2015.","short":"M.K. Hoang, J. Schmalenstroeer, R. Haeb-Umbach, in: 40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015), 2015.","ieee":"M. K. Hoang, J. Schmalenstroeer, and R. Haeb-Umbach, “Aligning training models with smartphone properties in WiFi fingerprinting based indoor localization,” 2015.","apa":"Hoang, M. K., Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2015). Aligning training models with smartphone properties in WiFi fingerprinting based indoor localization. <i>40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)</i>."},"publication":"40th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2015)","user_id":"460","_id":"11874","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2015/HoSchHa2015.pdf","open_access":"1"}],"date_updated":"2023-10-26T08:11:43Z","author":[{"last_name":"Hoang","first_name":"Manh Kha","full_name":"Hoang, Manh Kha"},{"last_name":"Schmalenstroeer","first_name":"Joerg","full_name":"Schmalenstroeer, Joerg","id":"460"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"title":"Aligning training models with smartphone properties in WiFi fingerprinting based indoor localization","year":"2015","status":"public"},{"user_id":"44006","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/ChPuHa2014.pdf"}],"_id":"11746","language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:08Z","status":"public","title":"Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR","year":"2014","author":[{"first_name":"Aleksej","last_name":"Chinaev","full_name":"Chinaev, Aleksej"},{"full_name":"Puels, Marc","first_name":"Marc","last_name":"Puels"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:27:27Z","abstract":[{"lang":"eng","text":" \"A method for nonstationary noise robust automatic speech recognition (ASR) is to first estimate the changing noise statistics and second clean up the features prior to recognition accordingly. Here, the first is accomplished by noise tracking in the spectral domain, while the second relies on Bayesian enhancement in the feature domain. In this way we take advantage of our recently proposed maximum a-posteriori based (MAP-B) noise power spectral density estimation algorithm, which is able to estimate the noise statistics even in time-frequency bins dominated by speech. We show that MAP-B noise tracking leads to an improved noise model estimate in the feature domain compared to estimating noise in speech absence periods only, if the bias resulting from the nonlinear transformation from the spectral to the feature domain is accounted for. Consequently, ASR results are improved, as is shown by experiments conducted on the Aurora IV database.\" "}],"related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/ChPuHa2014_Talk.pdf","relation":"supplementary_material","description":"Presentation"}]},"publication":"11. ITG Fachtagung Sprachkommunikation (ITG 2014)","citation":{"chicago":"Chinaev, Aleksej, Marc Puels, and Reinhold Haeb-Umbach. “Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR.” In <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","ama":"Chinaev A, Puels M, Haeb-Umbach R. Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR. In: <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>. ; 2014.","short":"A. Chinaev, M. Puels, R. Haeb-Umbach, in: 11. ITG Fachtagung Sprachkommunikation (ITG 2014), 2014.","bibtex":"@inproceedings{Chinaev_Puels_Haeb-Umbach_2014, title={Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR}, booktitle={11. ITG Fachtagung Sprachkommunikation (ITG 2014)}, author={Chinaev, Aleksej and Puels, Marc and Haeb-Umbach, Reinhold}, year={2014} }","mla":"Chinaev, Aleksej, et al. “Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR.” <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","apa":"Chinaev, A., Puels, M., &#38; Haeb-Umbach, R. (2014). Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR. In <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>.","ieee":"A. Chinaev, M. Puels, and R. Haeb-Umbach, “Spectral Noise Tracking for Improved Nonstationary Noise Robust ASR,” in <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014."}},{"abstract":[{"text":" \"In this contribution we derive a variational EM (VEM) algorithm for model selection in complex Watson mixture models, which have been recently proposed as a model of the distribution of normalized microphone array signals in the short-time Fourier transform domain. The VEM algorithm is applied to count the number of active sources in a speech mixture by iteratively estimating the mode vectors of the Watson distributions and suppressing the signals from the corresponding directions. A key theoretical contribution is the derivation of the MMSE estimate of a quadratic form involving the mode vector of the Watson distribution. The experimental results demonstrate the effectiveness of the source counting approach at moderately low SNR. It is further shown that the VEM algorithm is more robust w.r.t. used threshold values.\" ","lang":"eng"}],"related_material":{"link":[{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2014/DrChTrHa2014_Poster.pdf","description":"Poster"}]},"citation":{"mla":"Drude, Lukas, et al. “Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models.” <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","apa":"Drude, L., Chinaev, A., Tran Vu, D. H., &#38; Haeb-Umbach, R. (2014). Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models. In <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>.","ieee":"L. Drude, A. Chinaev, D. H. Tran Vu, and R. Haeb-Umbach, “Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models,” in <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","short":"L. Drude, A. Chinaev, D.H. Tran Vu, R. Haeb-Umbach, in: 39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014), 2014.","ama":"Drude L, Chinaev A, Tran Vu DH, Haeb-Umbach R. Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models. In: <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>. ; 2014.","chicago":"Drude, Lukas, Aleksej Chinaev, Dang Hai Tran Vu, and Reinhold Haeb-Umbach. “Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models.” In <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","bibtex":"@inproceedings{Drude_Chinaev_Tran Vu_Haeb-Umbach_2014, title={Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models}, booktitle={39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)}, author={Drude, Lukas and Chinaev, Aleksej and Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}, year={2014} }"},"publication":"39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)","oa":"1","department":[{"_id":"54"}],"type":"conference","date_created":"2019-07-12T05:27:34Z","date_updated":"2022-01-06T06:51:08Z","author":[{"id":"11213","full_name":"Drude, Lukas","last_name":"Drude","first_name":"Lukas"},{"full_name":"Chinaev, Aleksej","last_name":"Chinaev","first_name":"Aleksej"},{"full_name":"Tran Vu, Dang Hai","first_name":"Dang Hai","last_name":"Tran Vu"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}],"title":"Source Counting in Speech Mixtures Using a Variational EM Approach for Complexwatson Mixture Models","status":"public","year":"2014","user_id":"44006","_id":"11752","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/DrChTrHa2014.pdf"}]},{"user_id":"44006","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/DrChTrHaeb14.pdf"}],"page":"213-217","language":[{"iso":"eng"}],"_id":"11753","date_updated":"2022-01-06T06:51:08Z","title":"Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models","year":"2014","status":"public","author":[{"full_name":"Drude, Lukas","last_name":"Drude","first_name":"Lukas","id":"11213"},{"full_name":"Chinaev, Aleksej","first_name":"Aleksej","last_name":"Chinaev"},{"first_name":"Dang Hai","last_name":"Tran Vu","full_name":"Tran Vu, Dang Hai"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"keyword":["Accuracy","Acoustics","Estimation","Mathematical model","Soruce separation","Speech","Vectors","Bayes methods","Blind source separation","Directional statistics","Number of speakers","Speaker diarization"],"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:27:35Z","abstract":[{"text":"This contribution describes a step-wise source counting algorithm to determine the number of speakers in an offline scenario. Each speaker is identified by a variational expectation maximization (VEM) algorithm for complex Watson mixture models and therefore directly yields beamforming vectors for a subsequent speech separation process. An observation selection criterion is proposed which improves the robustness of the source counting in noise. The algorithm is compared to an alternative VEM approach with Gaussian mixture models based on directions of arrival and shown to deliver improved source counting accuracy. The article concludes by extending the offline algorithm towards a low-latency online estimation of the number of active sources from the streaming input data.","lang":"eng"}],"related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/DrChTrHaeb14_Poster.pdf","relation":"supplementary_material","description":"Poster"}]},"publication":"14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)","citation":{"short":"L. Drude, A. Chinaev, D.H. Tran Vu, R. Haeb-Umbach, in: 14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014), 2014, pp. 213–217.","chicago":"Drude, Lukas, Aleksej Chinaev, Dang Hai Tran Vu, and Reinhold Haeb-Umbach. “Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models.” In <i>14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)</i>, 213–17, 2014.","ieee":"L. Drude, A. Chinaev, D. H. Tran Vu, and R. Haeb-Umbach, “Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models,” in <i>14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)</i>, 2014, pp. 213–217.","apa":"Drude, L., Chinaev, A., Tran Vu, D. H., &#38; Haeb-Umbach, R. (2014). Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models. In <i>14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)</i> (pp. 213–217).","bibtex":"@inproceedings{Drude_Chinaev_Tran Vu_Haeb-Umbach_2014, title={Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models}, booktitle={14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)}, author={Drude, Lukas and Chinaev, Aleksej and Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}, year={2014}, pages={213–217} }","ama":"Drude L, Chinaev A, Tran Vu DH, Haeb-Umbach R. Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models. In: <i>14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)</i>. ; 2014:213-217.","mla":"Drude, Lukas, et al. “Towards Online Source Counting in Speech Mixtures Applying a Variational EM for Complex Watson Mixture Models.” <i>14th International Workshop on Acoustic Signal Enhancement (IWAENC 2014)</i>, 2014, pp. 213–17."}},{"date_created":"2019-07-12T05:28:46Z","type":"conference","oa":"1","department":[{"_id":"54"}],"publication":"39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)","citation":{"chicago":"Heymann, Jahn, Oliver Walter, Reinhold Haeb-Umbach, and Bhiksha Raj. “Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices.” In <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","short":"J. Heymann, O. Walter, R. Haeb-Umbach, B. Raj, in: 39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014), 2014.","apa":"Heymann, J., Walter, O., Haeb-Umbach, R., &#38; Raj, B. (2014). Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices. In <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>.","ieee":"J. Heymann, O. Walter, R. Haeb-Umbach, and B. Raj, “Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices,” in <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","ama":"Heymann J, Walter O, Haeb-Umbach R, Raj B. Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices. In: <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>. ; 2014.","bibtex":"@inproceedings{Heymann_Walter_Haeb-Umbach_Raj_2014, title={Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices}, booktitle={39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)}, author={Heymann, Jahn and Walter, Oliver and Haeb-Umbach, Reinhold and Raj, Bhiksha}, year={2014} }","mla":"Heymann, Jahn, et al. “Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices.” <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014."},"related_material":{"link":[{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2014/HeWaHa2014_Poster.pdf","description":"Poster"}]},"abstract":[{"text":" \"In this paper we present an algorithm for the unsupervised segmentation of a lattice produced by a phoneme recognizer into words. Using a lattice rather than a single phoneme string accounts for the uncertainty of the recognizer about the true label sequence. An example application is the discovery of lexical units from the output of an error-prone phoneme recognizer in a zero-resource setting, where neither the lexicon nor the language model (LM) is known. We propose a computationally efficient iterative approach, which alternates between the following two steps: First, the most probable string is extracted from the lattice using a phoneme LM learned on the segmentation result of the previous iteration. Second, word segmentation is performed on the extracted string using a word and phoneme LM which is learned alongside the new segmentation. We present results on lattices produced by a phoneme recognizer on the WSJCAM0 dataset. We show that our approach delivers superior segmentation performance than an earlier approach found in the literature, in particular for higher-order language models. \" ","lang":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/HeWaHa2014.pdf"}],"language":[{"iso":"eng"}],"_id":"11814","user_id":"44006","status":"public","year":"2014","title":"Iterative Bayesian Word Segmentation for Unspuervised Vocabulary Discovery from Phoneme Lattices","author":[{"id":"9168","last_name":"Heymann","first_name":"Jahn","full_name":"Heymann, Jahn"},{"last_name":"Walter","first_name":"Oliver","full_name":"Walter, Oliver"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"},{"last_name":"Raj","first_name":"Bhiksha","full_name":"Raj, Bhiksha"}],"date_updated":"2022-01-06T06:51:09Z"},{"author":[{"full_name":"Jacob, Florian","first_name":"Florian","last_name":"Jacob"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"title":"Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking","year":"2014","status":"public","date_updated":"2022-01-06T06:51:11Z","_id":"11831","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/JaHa2014.pdf","open_access":"1"}],"user_id":"44006","citation":{"mla":"Jacob, Florian, and Reinhold Haeb-Umbach. “Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking.” <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","bibtex":"@inproceedings{Jacob_Haeb-Umbach_2014, title={Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking}, booktitle={11. ITG Fachtagung Sprachkommunikation (ITG 2014)}, author={Jacob, Florian and Haeb-Umbach, Reinhold}, year={2014} }","ama":"Jacob F, Haeb-Umbach R. Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking. In: <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>. ; 2014.","ieee":"F. Jacob and R. Haeb-Umbach, “Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking,” in <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","apa":"Jacob, F., &#38; Haeb-Umbach, R. (2014). Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking. In <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>.","short":"F. Jacob, R. Haeb-Umbach, in: 11. ITG Fachtagung Sprachkommunikation (ITG 2014), 2014.","chicago":"Jacob, Florian, and Reinhold Haeb-Umbach. “Coordinate Mapping Between an Acoustic and Visual Sensor Network in the Shape Domain for a Joint Self-Calibrating Speaker Tracking.” In <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014."},"publication":"11. ITG Fachtagung Sprachkommunikation (ITG 2014)","related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/JaHa2014_Talk.pdf","relation":"supplementary_material","description":"Presentation"}]},"abstract":[{"lang":"eng","text":" \"Several self-localization algorithms have been proposed, that determine the positions of either acoustic or visual sensors autonomously. Usually these positions are given in a modality specific coordinate system, with an unknown rotation, translation and scale between the different systems. For a joint audiovisual tracking, where the different modalities support each other, the two modalities need to be mapped into a common coordinate system. In this paper we propose to estimate this mapping based on audiovisual correlates, i.e., a speaker that can be localized by both, a microphone and a camera network separately. The voice is tracked by a microphone network, which had to be calibrated by a self-localization algorithm at first, and the head is tracked by a calibrated camera network. Unlike existing Singular Value Decomposition based approaches to estimate the coordinate system mapping, we propose to perform an estimation in the shape domain, which turns out to be computationally more efficient. Simulations of the self-localization of an acoustic sensor network and a following coordinate mapping for a joint speaker localization showed a significant improvement of the localization performance, since the modalities were able to support each other.\" "}],"date_created":"2019-07-12T05:29:06Z","department":[{"_id":"54"}],"oa":"1","type":"conference"},{"citation":{"short":"V. Leutnant, A. Krueger, R. Haeb-Umbach, IEEE/ACM Transactions on Audio, Speech, and Language Processing 22 (2014) 95–109.","chicago":"Leutnant, Volker, Alexander Krueger, and Reinhold Haeb-Umbach. “A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i> 22, no. 1 (2014): 95–109. <a href=\"https://doi.org/10.1109/TASLP.2013.2285480\">https://doi.org/10.1109/TASLP.2013.2285480</a>.","ieee":"V. Leutnant, A. Krueger, and R. Haeb-Umbach, “A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech,” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, vol. 22, no. 1, pp. 95–109, 2014.","apa":"Leutnant, V., Krueger, A., &#38; Haeb-Umbach, R. (2014). A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, <i>22</i>(1), 95–109. <a href=\"https://doi.org/10.1109/TASLP.2013.2285480\">https://doi.org/10.1109/TASLP.2013.2285480</a>","bibtex":"@article{Leutnant_Krueger_Haeb-Umbach_2014, title={A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech}, volume={22}, DOI={<a href=\"https://doi.org/10.1109/TASLP.2013.2285480\">10.1109/TASLP.2013.2285480</a>}, number={1}, journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing}, author={Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}, year={2014}, pages={95–109} }","ama":"Leutnant V, Krueger A, Haeb-Umbach R. A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>. 2014;22(1):95-109. doi:<a href=\"https://doi.org/10.1109/TASLP.2013.2285480\">10.1109/TASLP.2013.2285480</a>","mla":"Leutnant, Volker, et al. “A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, vol. 22, no. 1, 2014, pp. 95–109, doi:<a href=\"https://doi.org/10.1109/TASLP.2013.2285480\">10.1109/TASLP.2013.2285480</a>."},"user_id":"44006","volume":22,"page":"95-109","_id":"11861","status":"public","keyword":["computational complexity","reverberation","speech recognition","automatic speech recognition","background noise","clean speech","computational complexity","energy compensation","logarithmic mel power spectral domain","mel frequency cepstral coefficients","microphone input signals","model-based feature compensation schemes","noisy reverberant speech automatic recognition","noisy reverberant speech features","reverberation","Atmospheric modeling","Computational modeling","Noise","Noise measurement","Reverberation","Speech","Vectors","Model-based feature compensation","observation model for reverberant and noisy speech","recursive observation model","robust automatic speech recognition"],"type":"journal_article","department":[{"_id":"54"}],"date_created":"2019-07-12T05:29:41Z","abstract":[{"text":"In this contribution we present a theoretical and experimental investigation into the effects of reverberation and noise on features in the logarithmic mel power spectral domain, an intermediate stage in the computation of the mel frequency cepstral coefficients, prevalent in automatic speech recognition (ASR). Gaining insight into the complex interaction between clean speech, noise, and noisy reverberant speech features is essential for any ASR system to be robust against noise and reverberation present in distant microphone input signals. The findings are gathered in a probabilistic formulation of an observation model which may be used in model-based feature compensation schemes. The proposed observation model extends previous models in three major directions: First, the contribution of additive background noise to the observation error is explicitly taken into account. Second, an energy compensation constant is introduced which ensures an unbiased estimate of the reverberant speech features, and, third, a recursive variant of the observation model is developed resulting in reduced computational complexity when used in model-based feature compensation. The experimental section is used to evaluate the accuracy of the model and to describe how its parameters can be determined from test data.","lang":"eng"}],"publication":"IEEE/ACM Transactions on Audio, Speech, and Language Processing","issue":"1","doi":"10.1109/TASLP.2013.2285480","language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:11Z","intvolume":"        22","year":"2014","title":"A New Observation Model in the Logarithmic Mel Power Spectral Domain for the Automatic Recognition of Noisy Reverberant Speech","author":[{"full_name":"Leutnant, Volker","last_name":"Leutnant","first_name":"Volker"},{"full_name":"Krueger, Alexander","last_name":"Krueger","first_name":"Alexander"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"publication_identifier":{"issn":["2329-9290"]}},{"volume":22,"user_id":"44006","_id":"11867","page":"745-777","status":"public","oa":"1","citation":{"apa":"Li, J., Deng, L., Gong, Y., &#38; Haeb-Umbach, R. (2014). An Overview of Noise-Robust Automatic Speech Recognition. <i>IEEE Transactions on Audio, Speech and Language Processing</i>, <i>22</i>(4), 745–777. <a href=\"https://doi.org/10.1109/TASLP.2014.2304637\">https://doi.org/10.1109/TASLP.2014.2304637</a>","ieee":"J. Li, L. Deng, Y. Gong, and R. Haeb-Umbach, “An Overview of Noise-Robust Automatic Speech Recognition,” <i>IEEE Transactions on Audio, Speech and Language Processing</i>, vol. 22, no. 4, pp. 745–777, 2014.","short":"J. Li, L. Deng, Y. Gong, R. Haeb-Umbach, IEEE Transactions on Audio, Speech and Language Processing 22 (2014) 745–777.","chicago":"Li, Jinyu, Li Deng, Yifan Gong, and Reinhold Haeb-Umbach. “An Overview of Noise-Robust Automatic Speech Recognition.” <i>IEEE Transactions on Audio, Speech and Language Processing</i> 22, no. 4 (2014): 745–77. <a href=\"https://doi.org/10.1109/TASLP.2014.2304637\">https://doi.org/10.1109/TASLP.2014.2304637</a>.","mla":"Li, Jinyu, et al. “An Overview of Noise-Robust Automatic Speech Recognition.” <i>IEEE Transactions on Audio, Speech and Language Processing</i>, vol. 22, no. 4, 2014, pp. 745–77, doi:<a href=\"https://doi.org/10.1109/TASLP.2014.2304637\">10.1109/TASLP.2014.2304637</a>.","ama":"Li J, Deng L, Gong Y, Haeb-Umbach R. An Overview of Noise-Robust Automatic Speech Recognition. <i>IEEE Transactions on Audio, Speech and Language Processing</i>. 2014;22(4):745-777. doi:<a href=\"https://doi.org/10.1109/TASLP.2014.2304637\">10.1109/TASLP.2014.2304637</a>","bibtex":"@article{Li_Deng_Gong_Haeb-Umbach_2014, title={An Overview of Noise-Robust Automatic Speech Recognition}, volume={22}, DOI={<a href=\"https://doi.org/10.1109/TASLP.2014.2304637\">10.1109/TASLP.2014.2304637</a>}, number={4}, journal={IEEE Transactions on Audio, Speech and Language Processing}, author={Li, Jinyu and Deng, Li and Gong, Yifan and Haeb-Umbach, Reinhold}, year={2014}, pages={745–777} }"},"doi":"10.1109/TASLP.2014.2304637","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"http://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=6732927"}],"intvolume":"        22","date_updated":"2022-01-06T06:51:11Z","author":[{"last_name":"Li","first_name":"Jinyu","full_name":"Li, Jinyu"},{"last_name":"Deng","first_name":"Li","full_name":"Deng, Li"},{"first_name":"Yifan","last_name":"Gong","full_name":"Gong, Yifan"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"title":"An Overview of Noise-Robust Automatic Speech Recognition","year":"2014","department":[{"_id":"54"}],"type":"journal_article","keyword":["Speech recognition","compensation","distortion modeling","joint model training","noise","robustness","uncertainty processing"],"date_created":"2019-07-12T05:29:47Z","abstract":[{"text":"New waves of consumer-centric applications, such as voice search and voice interaction with mobile devices and home entertainment systems, increasingly require automatic speech recognition (ASR) to be robust to the full range of real-world noise and other acoustic distorting conditions. Despite its practical importance, however, the inherent links between and distinctions among the myriad of methods for noise-robust ASR have yet to be carefully studied in order to advance the field further. To this end, it is critical to establish a solid, consistent, and common mathematical foundation for noise-robust ASR, which is lacking at present. This article is intended to fill this gap and to provide a thorough overview of modern noise-robust techniques for ASR developed over the past 30 years. We emphasize methods that are proven to be successful and that are likely to sustain or expand their future applicability. We distill key insights from our comprehensive overview in this field and take a fresh look at a few old problems, which nevertheless are still highly relevant today. Specifically, we have analyzed and categorized a wide range of noise-robust techniques using five different criteria: 1) feature-domain vs. model-domain processing, 2) the use of prior knowledge about the acoustic environment distortion, 3) the use of explicit environment-distortion models, 4) deterministic vs. uncertainty processing, and 5) the use of acoustic models trained jointly with the same feature enhancement or model adaptation process used in the testing stage. With this taxonomy-oriented review, we equip the reader with the insight to choose among techniques and with the awareness of the performance-complexity tradeoffs. The pros and cons of using different noise-robust ASR techniques in practical application scenarios are provided as a guide to interested practitioners. The current challenges and future research directions in this field is also carefully analyzed.","lang":"eng"}],"issue":"4","publication":"IEEE Transactions on Audio, Speech and Language Processing"},{"type":"conference","oa":"1","department":[{"_id":"54"}],"date_created":"2019-07-12T05:30:46Z","abstract":[{"lang":"eng","text":"In this paper, we investigate unsupervised acoustic model training approaches for dysarthric-speech recognition. These models are first, frame-based Gaussian posteriorgrams, obtained from Vector Quantization (VQ), second, so-called Acoustic Unit Descriptors (AUDs), which are hidden Markov models of phone-like units, that are trained in an unsupervised fashion, and, third, posteriorgrams computed on the AUDs. Experiments were carried out on a database collected from a home automation task and containing nine speakers, of which seven are considered to utter dysarthric speech. All unsupervised modeling approaches delivered significantly better recognition rates than a speaker-independent phoneme recognition baseline, showing the suitability of unsupervised acoustic model training for dysarthric speech. While the AUD models led to the most compact representation of an utterance for the subsequent semantic inference stage, posteriorgram-based representations resulted in higher recognition rates, with the Gaussian posteriorgram achieving the highest slot filling F-score of 97.02%. Index Terms: unsupervised learning, acoustic unit descriptors, dysarthric speech, non-negative matrix factorization"}],"related_material":{"link":[{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2014/WaDeHaebGeOnVa14_Poster.pdf","description":"Poster"},{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2014/WaDeHaebGeOnVa14_Spotlight.pdf","description":"Spotlight"}]},"publication":"INTERSPEECH 2014","citation":{"mla":"Walter, Oliver, et al. “An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface.” <i>INTERSPEECH 2014</i>, 2014.","ama":"Walter O, Despotovic V, Haeb-Umbach R, Gemmeke J, Ons B, Van hamme H. An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface. In: <i>INTERSPEECH 2014</i>. ; 2014.","bibtex":"@inproceedings{Walter_Despotovic_Haeb-Umbach_Gemmeke_Ons_Van hamme_2014, title={An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface}, booktitle={INTERSPEECH 2014}, author={Walter, Oliver and Despotovic, Vladimir and Haeb-Umbach, Reinhold and Gemmeke, Jrt and Ons, Bart and Van hamme, Hugo}, year={2014} }","apa":"Walter, O., Despotovic, V., Haeb-Umbach, R., Gemmeke, J., Ons, B., &#38; Van hamme, H. (2014). An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface. In <i>INTERSPEECH 2014</i>.","ieee":"O. Walter, V. Despotovic, R. Haeb-Umbach, J. Gemmeke, B. Ons, and H. Van hamme, “An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface,” in <i>INTERSPEECH 2014</i>, 2014.","chicago":"Walter, Oliver, Vladimir Despotovic, Reinhold Haeb-Umbach, Jrt Gemmeke, Bart Ons, and Hugo Van hamme. “An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface.” In <i>INTERSPEECH 2014</i>, 2014.","short":"O. Walter, V. Despotovic, R. Haeb-Umbach, J. Gemmeke, B. Ons, H. Van hamme, in: INTERSPEECH 2014, 2014."},"user_id":"44006","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/WaDeHaebGeOnVa14.pdf","open_access":"1"}],"_id":"11918","language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:12Z","status":"public","title":"An Evaluation of Unsupervised Acoustic Model Training for a Dysarthric Speech Interface","year":"2014","author":[{"full_name":"Walter, Oliver","last_name":"Walter","first_name":"Oliver"},{"first_name":"Vladimir","last_name":"Despotovic","full_name":"Despotovic, Vladimir"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"},{"full_name":"Gemmeke, Jrt","last_name":"Gemmeke","first_name":"Jrt"},{"full_name":"Ons, Bart","first_name":"Bart","last_name":"Ons"},{"last_name":"Van hamme","first_name":"Hugo","full_name":"Van hamme, Hugo"}]},{"author":[{"id":"460","last_name":"Schmalenstroeer","first_name":"Joerg","full_name":"Schmalenstroeer, Joerg"},{"full_name":"Jebramcik, Patrick","first_name":"Patrick","last_name":"Jebramcik"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"publication_identifier":{"issn":["0165-1684"]},"year":"2014","title":"A combined hardware-software approach for acoustic sensor network synchronization ","date_updated":"2023-10-26T08:11:22Z","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"http://www.sciencedirect.com/science/article/pii/S0165168414002990"}],"doi":"http://dx.doi.org/10.1016/j.sigpro.2014.06.030","issue":"0","publication":"Signal Processing","abstract":[{"lang":"eng","text":"Abstract In this paper we present an approach for synchronizing a wireless acoustic sensor network using a two-stage procedure. First the clock frequency and phase differences between pairs of nodes are estimated employing a two-way message exchange protocol. The estimates are further improved in a Kalman filter with a dedicated observation error model. In the second stage network-wide synchronization is achieved by means of a gossiping algorithm which estimates the average clock frequency and phase of the sensor nodes. These averages are viewed as frequency and phase of a virtual master clock, to which the clocks of the sensor nodes have to be adjusted. The amount of adjustment is computed in a specific control loop. While these steps are done in software, the actual sampling rate correction is carried out in hardware by using an adjustable frequency synthesizer. Experimental results obtained from hardware devices and software simulations of large scale networks are presented."}],"date_created":"2019-07-12T05:30:23Z","department":[{"_id":"54"}],"type":"journal_article","keyword":["Gossip algorithm"],"status":"public","_id":"11898","page":" - ","user_id":"460","citation":{"mla":"Schmalenstroeer, Joerg, et al. “A Combined Hardware-Software Approach for Acoustic Sensor Network Synchronization .” <i>Signal Processing</i>, no. 0, 2014, p., doi:<a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>.","bibtex":"@article{Schmalenstroeer_Jebramcik_Haeb-Umbach_2014, title={A combined hardware-software approach for acoustic sensor network synchronization }, DOI={<a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>}, number={0}, journal={Signal Processing}, author={Schmalenstroeer, Joerg and Jebramcik, Patrick and Haeb-Umbach, Reinhold}, year={2014} }","ama":"Schmalenstroeer J, Jebramcik P, Haeb-Umbach R. A combined hardware-software approach for acoustic sensor network synchronization . <i>Signal Processing</i>. 2014;(0). doi:<a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>","ieee":"J. Schmalenstroeer, P. Jebramcik, and R. Haeb-Umbach, “A combined hardware-software approach for acoustic sensor network synchronization ,” <i>Signal Processing</i>, no. 0, p., 2014, doi: <a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>.","apa":"Schmalenstroeer, J., Jebramcik, P., &#38; Haeb-Umbach, R. (2014). A combined hardware-software approach for acoustic sensor network synchronization . <i>Signal Processing</i>, <i>0</i>. <a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>","short":"J. Schmalenstroeer, P. Jebramcik, R. Haeb-Umbach, Signal Processing (2014).","chicago":"Schmalenstroeer, Joerg, Patrick Jebramcik, and Reinhold Haeb-Umbach. “A Combined Hardware-Software Approach for Acoustic Sensor Network Synchronization .” <i>Signal Processing</i>, no. 0 (2014). <a href=\"http://dx.doi.org/10.1016/j.sigpro.2014.06.030\">http://dx.doi.org/10.1016/j.sigpro.2014.06.030</a>."},"quality_controlled":"1","oa":"1"},{"date_created":"2019-07-12T05:30:22Z","department":[{"_id":"54"}],"oa":"1","type":"conference","citation":{"ieee":"J. Schmalenstroeer, P. Jebramcik, and R. Haeb-Umbach, “A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks,” 2014.","apa":"Schmalenstroeer, J., Jebramcik, P., &#38; Haeb-Umbach, R. (2014). A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks. <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>.","chicago":"Schmalenstroeer, Joerg, Patrick Jebramcik, and Reinhold Haeb-Umbach. “A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks.” In <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","short":"J. Schmalenstroeer, P. Jebramcik, R. Haeb-Umbach, in: 39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014), 2014.","mla":"Schmalenstroeer, Joerg, et al. “A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks.” <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>, 2014.","bibtex":"@inproceedings{Schmalenstroeer_Jebramcik_Haeb-Umbach_2014, title={A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks}, booktitle={39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)}, author={Schmalenstroeer, Joerg and Jebramcik, Patrick and Haeb-Umbach, Reinhold}, year={2014} }","ama":"Schmalenstroeer J, Jebramcik P, Haeb-Umbach R. A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks. In: <i>39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)</i>. ; 2014."},"publication":"39th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2014)","quality_controlled":"1","abstract":[{"text":" \"In this paper we present an approach for synchronizing the sampling clocks of distributed microphones over a wireless network. The proposed system uses a two stage procedure. It first employs a two-way message exchange algorithm to estimate the clock phase and frequency difference between two nodes and then uses a gossiping algorithmto estimate a virtual master clock, to which all sensor nodes synchronize. Simulation results are presented for networks of different topology and size, showing the effectiveness of our approach.\" ","lang":"eng"}],"related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/SchHaebICASSP2014_Poster.pdf","relation":"supplementary_material","description":"Poster"}]},"language":[{"iso":"eng"}],"_id":"11897","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/SchHae2014.pdf"}],"user_id":"460","author":[{"id":"460","full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer"},{"last_name":"Jebramcik","first_name":"Patrick","full_name":"Jebramcik, Patrick"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"year":"2014","status":"public","title":"A Gossiping Approach to Sampling Clock Synchronization in Wireless Acoustic Sensor Networks","date_updated":"2023-10-26T08:11:31Z"},{"date_updated":"2023-10-26T08:14:00Z","author":[{"full_name":"Schmalenstroeer, Joerg","last_name":"Schmalenstroeer","first_name":"Joerg","id":"460"},{"full_name":"Zhao, Weile","last_name":"Zhao","first_name":"Weile"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"year":"2014","status":"public","title":"Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization","user_id":"460","_id":"11903","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2014/SchHaebITG2014.pdf"}],"quality_controlled":"1","abstract":[{"lang":"eng","text":"\"Acoustic sensor network clock synchronization via time stamp exchange between the sensor nodes is not accurate enough for many acoustic signal processing tasks, such as speaker localization. To improve synchronization accuracy it has therefore been proposed to employ a Kalman Filter to obtain improved frequency deviation and phase offset estimates. The estimation requires a statistical model of the errors of the measurements obtained from the time stamp exchange algorithm. These errors are caused by random transmission delays and hardware effects and are thus network specific. In this contribution we develop an algorithm to estimate the parameters of the measurement error model alongside the Kalman filter based sampling clock synchronization, employing the Expectation Maximization algorithm. Simulation results demonstrate that the online estimation of the error model parameters leads only to a small degradation of the synchronization performance compared to a perfectly known observation error model.\""}],"related_material":{"link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2014/SchHaebITG2014_Poster.pdf","relation":"supplementary_material","description":"Poster"},{"relation":"supplementary_material","url":"https://groups.uni-paderborn.de/nt/pubs/2014/SchHaebITG2014_Demo.pdf","description":"Demo"}]},"citation":{"mla":"Schmalenstroeer, Joerg, et al. “Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization.” <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","ama":"Schmalenstroeer J, Zhao W, Haeb-Umbach R. Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization. In: <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>. ; 2014.","bibtex":"@inproceedings{Schmalenstroeer_Zhao_Haeb-Umbach_2014, title={Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization}, booktitle={11. ITG Fachtagung Sprachkommunikation (ITG 2014)}, author={Schmalenstroeer, Joerg and Zhao, Weile and Haeb-Umbach, Reinhold}, year={2014} }","apa":"Schmalenstroeer, J., Zhao, W., &#38; Haeb-Umbach, R. (2014). Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization. <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>.","ieee":"J. Schmalenstroeer, W. Zhao, and R. Haeb-Umbach, “Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization,” 2014.","chicago":"Schmalenstroeer, Joerg, Weile Zhao, and Reinhold Haeb-Umbach. “Online Observation Error Model Estimation for Acoustic Sensor Network Synchronization.” In <i>11. ITG Fachtagung Sprachkommunikation (ITG 2014)</i>, 2014.","short":"J. Schmalenstroeer, W. Zhao, R. Haeb-Umbach, in: 11. ITG Fachtagung Sprachkommunikation (ITG 2014), 2014."},"publication":"11. ITG Fachtagung Sprachkommunikation (ITG 2014)","oa":"1","department":[{"_id":"54"}],"type":"conference","date_created":"2019-07-12T05:30:29Z"}]
