[{"language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2011/KrWaHa11.pdf","open_access":"1"}],"doi":"10.1109/TASL.2010.2047324","author":[{"full_name":"Krueger, Alexander","last_name":"Krueger","first_name":"Alexander"},{"first_name":"Ernst","last_name":"Warsitz","full_name":"Warsitz, Ernst"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"year":"2011","title":"Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation","intvolume":"        19","date_updated":"2022-01-06T06:51:11Z","date_created":"2019-07-12T05:29:28Z","department":[{"_id":"54"}],"keyword":["acoustical transfer function ratio","adaptive eigenvector tracking","array signal processing","beamformer design","blocking matrix","eigenvalues and eigenfunctions","eigenvector-based transfer function ratios estimation","generalized sidelobe canceler","interference reduction","iterative methods","power iteration method","reduced speech distortions","reverberant enclosure","reverberation","speech enhancement","stationary noise"],"type":"journal_article","publication":"IEEE Transactions on Audio, Speech, and Language Processing","issue":"1","abstract":[{"lang":"eng","text":"In this paper, we present a novel blocking matrix and fixed beamformer design for a generalized sidelobe canceler for speech enhancement in a reverberant enclosure. They are based on a new method for estimating the acoustical transfer function ratios in the presence of stationary noise. The estimation method relies on solving a generalized eigenvalue problem in each frequency bin. An adaptive eigenvector tracking utilizing the power iteration method is employed and shown to achieve a high convergence speed. Simulation results demonstrate that the proposed beamformer leads to better noise and interference reduction and reduced speech distortions compared to other blocking matrix designs from the literature."}],"_id":"11850","page":"206-219","volume":19,"user_id":"44006","status":"public","oa":"1","citation":{"ieee":"A. Krueger, E. Warsitz, and R. Haeb-Umbach, “Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation,” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 19, no. 1, pp. 206–219, 2011.","apa":"Krueger, A., Warsitz, E., &#38; Haeb-Umbach, R. (2011). Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, <i>19</i>(1), 206–219. <a href=\"https://doi.org/10.1109/TASL.2010.2047324\">https://doi.org/10.1109/TASL.2010.2047324</a>","short":"A. Krueger, E. Warsitz, R. Haeb-Umbach, IEEE Transactions on Audio, Speech, and Language Processing 19 (2011) 206–219.","chicago":"Krueger, Alexander, Ernst Warsitz, and Reinhold Haeb-Umbach. “Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i> 19, no. 1 (2011): 206–19. <a href=\"https://doi.org/10.1109/TASL.2010.2047324\">https://doi.org/10.1109/TASL.2010.2047324</a>.","mla":"Krueger, Alexander, et al. “Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 19, no. 1, 2011, pp. 206–19, doi:<a href=\"https://doi.org/10.1109/TASL.2010.2047324\">10.1109/TASL.2010.2047324</a>.","bibtex":"@article{Krueger_Warsitz_Haeb-Umbach_2011, title={Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation}, volume={19}, DOI={<a href=\"https://doi.org/10.1109/TASL.2010.2047324\">10.1109/TASL.2010.2047324</a>}, number={1}, journal={IEEE Transactions on Audio, Speech, and Language Processing}, author={Krueger, Alexander and Warsitz, Ernst and Haeb-Umbach, Reinhold}, year={2011}, pages={206–219} }","ama":"Krueger A, Warsitz E, Haeb-Umbach R. Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>. 2011;19(1):206-219. doi:<a href=\"https://doi.org/10.1109/TASL.2010.2047324\">10.1109/TASL.2010.2047324</a>"}},{"citation":{"chicago":"Fischer, Kerstin, Kilian Foth, Katharina Rohlfing, and Britta Wrede. “Mindful Tutors: Linguistic Choice and Action Demonstration in Speech to Infants and a Simulated Robot.” <i>Interaction Studies</i> 12, no. 1 (2011): 134–61. <a href=\"https://doi.org/10.1075/is.12.1.06fis\">https://doi.org/10.1075/is.12.1.06fis</a>.","short":"K. Fischer, K. Foth, K. Rohlfing, B. Wrede, Interaction Studies 12 (2011) 134–161.","ieee":"K. Fischer, K. Foth, K. Rohlfing, and B. Wrede, “Mindful tutors: Linguistic choice and action demonstration in speech to infants and a simulated robot,” <i>Interaction Studies</i>, vol. 12, no. 1, pp. 134–161, 2011, doi: <a href=\"https://doi.org/10.1075/is.12.1.06fis\">10.1075/is.12.1.06fis</a>.","apa":"Fischer, K., Foth, K., Rohlfing, K., &#38; Wrede, B. (2011). Mindful tutors: Linguistic choice and action demonstration in speech to infants and a simulated robot. <i>Interaction Studies</i>, <i>12</i>(1), 134–161. <a href=\"https://doi.org/10.1075/is.12.1.06fis\">https://doi.org/10.1075/is.12.1.06fis</a>","bibtex":"@article{Fischer_Foth_Rohlfing_Wrede_2011, title={Mindful tutors: Linguistic choice and action demonstration in speech to infants and a simulated robot}, volume={12}, DOI={<a href=\"https://doi.org/10.1075/is.12.1.06fis\">10.1075/is.12.1.06fis</a>}, number={1}, journal={Interaction Studies}, publisher={John Benjamins Publishing Company}, author={Fischer, Kerstin and Foth, Kilian and Rohlfing, Katharina and Wrede, Britta}, year={2011}, pages={134–161} }","ama":"Fischer K, Foth K, Rohlfing K, Wrede B. Mindful tutors: Linguistic choice and action demonstration in speech to infants and a simulated robot. <i>Interaction Studies</i>. 2011;12(1):134-161. doi:<a href=\"https://doi.org/10.1075/is.12.1.06fis\">10.1075/is.12.1.06fis</a>","mla":"Fischer, Kerstin, et al. “Mindful Tutors: Linguistic Choice and Action Demonstration in Speech to Infants and a Simulated Robot.” <i>Interaction Studies</i>, vol. 12, no. 1, John Benjamins Publishing Company, 2011, pp. 134–61, doi:<a href=\"https://doi.org/10.1075/is.12.1.06fis\">10.1075/is.12.1.06fis</a>."},"status":"public","user_id":"14931","volume":12,"page":"134-161","_id":"17233","publisher":"John Benjamins Publishing Company","abstract":[{"lang":"eng","text":"It has been proposed that the design of robots might benefit from interactions that are similar to caregiver–child interactions, which is tailored to children’s respective capacities to a high degree. However, so far little is known about how people adapt their tutoring behaviour to robots and whether robots can evoke input that is similar to child-directed interaction. The paper presents detailed analyses of speakers’ linguistic and non-linguistic behaviour, such as action demonstration, in two comparable situations: In one experiment, parents described and explained to their nonverbal infants the use of certain everyday objects; in the other experiment, participants tutored a simulated robot on the same objects. The results, which show considerable differences between the two situations on almost all measures, are discussed in the light of the computer-as-social-actor paradigm and the register hypothesis."}],"publication":"Interaction Studies","issue":"1","type":"journal_article","keyword":["human–robot interaction (HRI)","social communication","register theory","motionese","robotese","child-directed speech (CDS)","motherese","mindless transfer","computers-as-social-actors"],"department":[{"_id":"749"}],"date_created":"2020-06-24T13:01:57Z","date_updated":"2023-02-01T12:56:04Z","intvolume":"        12","year":"2011","title":"Mindful tutors: Linguistic choice and action demonstration in speech to infants and a simulated robot","publication_identifier":{"issn":["1572-0381"]},"author":[{"full_name":"Fischer, Kerstin","last_name":"Fischer","first_name":"Kerstin"},{"last_name":"Foth","first_name":"Kilian","full_name":"Foth, Kilian"},{"id":"50352","full_name":"Rohlfing, Katharina","last_name":"Rohlfing","first_name":"Katharina"},{"first_name":"Britta","last_name":"Wrede","full_name":"Wrede, Britta"}],"doi":"10.1075/is.12.1.06fis","language":[{"iso":"eng"}]},{"issue":"7","publication":"IEEE Transactions on Audio, Speech, and Language Processing","abstract":[{"text":"In this paper, we present a new technique for automatic speech recognition (ASR) in reverberant environments. Our approach is aimed at the enhancement of the logarithmic Mel power spectrum, which is computed at an intermediate stage to obtain the widely used Mel frequency cepstral coefficients (MFCCs). Given the reverberant logarithmic Mel power spectral coefficients (LMPSCs), a minimum mean square error estimate of the clean LMPSCs is computed by carrying out Bayesian inference. We employ switching linear dynamical models as an a priori model for the dynamics of the clean LMPSCs. Further, we derive a stochastic observation model which relates the clean to the reverberant LMPSCs through a simplified model of the room impulse response (RIR). This model requires only two parameters, namely RIR energy and reverberation time, which can be estimated from the captured microphone signal. The performance of the proposed enhancement technique is studied on the AURORA5 database and compared to that of constrained maximum-likelihood linear regression (CMLLR). It is shown by experimental results that our approach significantly outperforms CMLLR and that up to 80\\% of the errors caused by the reverberation are recovered. In addition to the fact that the approach is compatible with the standard MFCC feature vectors, it leaves the ASR back-end unchanged. It is of moderate computational complexity and suitable for real time applications.","lang":"eng"}],"date_created":"2019-07-12T05:29:23Z","department":[{"_id":"54"}],"type":"journal_article","keyword":["ASR","AURORA5 database","automatic speech recognition","Bayesian inference","belief networks","CMLLR","computational complexity","constrained maximum likelihood linear regression","least mean squares methods","LMPSC computation","logarithmic Mel power spectrum","maximum likelihood estimation","Mel frequency cepstral coefficients","MFCC feature vectors","microphone signal","minimum mean square error estimation","model-based feature enhancement","regression analysis","reverberant speech recognition","reverberation","RIR energy","room impulse response","speech recognition","stochastic observation model","stochastic processes"],"author":[{"last_name":"Krueger","first_name":"Alexander","full_name":"Krueger, Alexander"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"title":"Model-Based Feature Enhancement for Reverberant Speech Recognition","year":"2010","intvolume":"        18","date_updated":"2022-01-06T06:51:11Z","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2010/KrHa10.pdf","open_access":"1"}],"doi":"10.1109/TASL.2010.2049684","citation":{"ieee":"A. Krueger and R. Haeb-Umbach, “Model-Based Feature Enhancement for Reverberant Speech Recognition,” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 18, no. 7, pp. 1692–1707, 2010.","apa":"Krueger, A., &#38; Haeb-Umbach, R. (2010). Model-Based Feature Enhancement for Reverberant Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, <i>18</i>(7), 1692–1707. <a href=\"https://doi.org/10.1109/TASL.2010.2049684\">https://doi.org/10.1109/TASL.2010.2049684</a>","chicago":"Krueger, Alexander, and Reinhold Haeb-Umbach. “Model-Based Feature Enhancement for Reverberant Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i> 18, no. 7 (2010): 1692–1707. <a href=\"https://doi.org/10.1109/TASL.2010.2049684\">https://doi.org/10.1109/TASL.2010.2049684</a>.","short":"A. Krueger, R. Haeb-Umbach, IEEE Transactions on Audio, Speech, and Language Processing 18 (2010) 1692–1707.","mla":"Krueger, Alexander, and Reinhold Haeb-Umbach. “Model-Based Feature Enhancement for Reverberant Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 18, no. 7, 2010, pp. 1692–707, doi:<a href=\"https://doi.org/10.1109/TASL.2010.2049684\">10.1109/TASL.2010.2049684</a>.","bibtex":"@article{Krueger_Haeb-Umbach_2010, title={Model-Based Feature Enhancement for Reverberant Speech Recognition}, volume={18}, DOI={<a href=\"https://doi.org/10.1109/TASL.2010.2049684\">10.1109/TASL.2010.2049684</a>}, number={7}, journal={IEEE Transactions on Audio, Speech, and Language Processing}, author={Krueger, Alexander and Haeb-Umbach, Reinhold}, year={2010}, pages={1692–1707} }","ama":"Krueger A, Haeb-Umbach R. Model-Based Feature Enhancement for Reverberant Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>. 2010;18(7):1692-1707. doi:<a href=\"https://doi.org/10.1109/TASL.2010.2049684\">10.1109/TASL.2010.2049684</a>"},"oa":"1","status":"public","_id":"11846","page":"1692-1707","volume":18,"user_id":"44006"},{"date_updated":"2022-01-06T06:51:12Z","year":"2010","title":"Blind speech separation employing directional statistics in an Expectation Maximization framework","status":"public","author":[{"last_name":"Tran Vu","first_name":"Dang Hai","full_name":"Tran Vu, Dang Hai"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"user_id":"44006","doi":"10.1109/ICASSP.2010.5495994","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2010/DaHa10-2.pdf","open_access":"1"}],"page":"241-244","_id":"11913","language":[{"iso":"eng"}],"abstract":[{"lang":"eng","text":"In this paper we propose to employ directional statistics in a complex vector space to approach the problem of blind speech separation in the presence of spatially correlated noise. We interpret the values of the short time Fourier transform of the microphone signals to be draws from a mixture of complex Watson distributions, a probabilistic model which naturally accounts for spatial aliasing. The parameters of the density are related to the a priori source probabilities, the power of the sources and the transfer function ratios from sources to sensors. Estimation formulas are derived for these parameters by employing the Expectation Maximization (EM) algorithm. The E-step corresponds to the estimation of the source presence probabilities for each time-frequency bin, while the M-step leads to a maximum signal-to-noise ratio (MaxSNR) beamformer in the presence of uncertainty about the source activity. Experimental results are reported for an implementation in a generalized sidelobe canceller (GSC) like spatial beamforming configuration for 3 speech sources with significant coherent noise in reverberant environments, demonstrating the usefulness of the novel modeling framework."}],"publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)","citation":{"short":"D.H. Tran Vu, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010), 2010, pp. 241–244.","chicago":"Tran Vu, Dang Hai, and Reinhold Haeb-Umbach. “Blind Speech Separation Employing Directional Statistics in an Expectation Maximization Framework.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)</i>, 241–44, 2010. <a href=\"https://doi.org/10.1109/ICASSP.2010.5495994\">https://doi.org/10.1109/ICASSP.2010.5495994</a>.","apa":"Tran Vu, D. H., &#38; Haeb-Umbach, R. (2010). Blind speech separation employing directional statistics in an Expectation Maximization framework. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)</i> (pp. 241–244). <a href=\"https://doi.org/10.1109/ICASSP.2010.5495994\">https://doi.org/10.1109/ICASSP.2010.5495994</a>","ieee":"D. H. Tran Vu and R. Haeb-Umbach, “Blind speech separation employing directional statistics in an Expectation Maximization framework,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)</i>, 2010, pp. 241–244.","ama":"Tran Vu DH, Haeb-Umbach R. Blind speech separation employing directional statistics in an Expectation Maximization framework. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)</i>. ; 2010:241-244. doi:<a href=\"https://doi.org/10.1109/ICASSP.2010.5495994\">10.1109/ICASSP.2010.5495994</a>","bibtex":"@inproceedings{Tran Vu_Haeb-Umbach_2010, title={Blind speech separation employing directional statistics in an Expectation Maximization framework}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2010.5495994\">10.1109/ICASSP.2010.5495994</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)}, author={Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}, year={2010}, pages={241–244} }","mla":"Tran Vu, Dang Hai, and Reinhold Haeb-Umbach. “Blind Speech Separation Employing Directional Statistics in an Expectation Maximization Framework.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)</i>, 2010, pp. 241–44, doi:<a href=\"https://doi.org/10.1109/ICASSP.2010.5495994\">10.1109/ICASSP.2010.5495994</a>."},"type":"conference","keyword":["array signal processing","blind source separation","blind speech separation","complex vector space","complex Watson distribution","directional statistics","expectation-maximisation algorithm","expectation maximization algorithm","Fourier transform","Fourier transforms","generalized sidelobe canceller","interference suppression","maximum signal-to-noise ratio beamformer","microphone signal","probabilistic model","spatial aliasing","spatial beamforming configuration","speech enhancement","statistical distributions"],"department":[{"_id":"54"}],"oa":"1","date_created":"2019-07-12T05:30:40Z"},{"status":"public","page":"845-856","_id":"11892","user_id":"460","volume":4,"citation":{"chicago":"Schmalenstroeer, Joerg, and Reinhold Haeb-Umbach. “Online Diarization of Streaming Audio-Visual Data for Smart Environments.” <i>IEEE Journal of Selected Topics in Signal Processing</i> 4, no. 5 (2010): 845–56. <a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">https://doi.org/10.1109/JSTSP.2010.2050519</a>.","short":"J. Schmalenstroeer, R. Haeb-Umbach, IEEE Journal of Selected Topics in Signal Processing 4 (2010) 845–856.","apa":"Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2010). Online Diarization of Streaming Audio-Visual Data for Smart Environments. <i>IEEE Journal of Selected Topics in Signal Processing</i>, <i>4</i>(5), 845–856. <a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">https://doi.org/10.1109/JSTSP.2010.2050519</a>","ieee":"J. Schmalenstroeer and R. Haeb-Umbach, “Online Diarization of Streaming Audio-Visual Data for Smart Environments,” <i>IEEE Journal of Selected Topics in Signal Processing</i>, vol. 4, no. 5, pp. 845–856, 2010, doi: <a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">10.1109/JSTSP.2010.2050519</a>.","ama":"Schmalenstroeer J, Haeb-Umbach R. Online Diarization of Streaming Audio-Visual Data for Smart Environments. <i>IEEE Journal of Selected Topics in Signal Processing</i>. 2010;4(5):845-856. doi:<a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">10.1109/JSTSP.2010.2050519</a>","bibtex":"@article{Schmalenstroeer_Haeb-Umbach_2010, title={Online Diarization of Streaming Audio-Visual Data for Smart Environments}, volume={4}, DOI={<a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">10.1109/JSTSP.2010.2050519</a>}, number={5}, journal={IEEE Journal of Selected Topics in Signal Processing}, author={Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2010}, pages={845–856} }","mla":"Schmalenstroeer, Joerg, and Reinhold Haeb-Umbach. “Online Diarization of Streaming Audio-Visual Data for Smart Environments.” <i>IEEE Journal of Selected Topics in Signal Processing</i>, vol. 4, no. 5, 2010, pp. 845–56, doi:<a href=\"https://doi.org/10.1109/JSTSP.2010.2050519\">10.1109/JSTSP.2010.2050519</a>."},"quality_controlled":"1","oa":"1","year":"2010","title":"Online Diarization of Streaming Audio-Visual Data for Smart Environments","author":[{"id":"460","full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"date_updated":"2023-10-26T08:10:18Z","intvolume":"         4","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2010/ScHa10.pdf","open_access":"1"}],"language":[{"iso":"eng"}],"doi":"10.1109/JSTSP.2010.2050519","issue":"5","publication":"IEEE Journal of Selected Topics in Signal Processing","abstract":[{"lang":"eng","text":"For an environment to be perceived as being smart, contextual information has to be gathered to adapt the system's behavior and its interface towards the user. Being a rich source of context information speech can be acquired unobtrusively by microphone arrays and then processed to extract information about the user and his environment. In this paper, a system for joint temporal segmentation, speaker localization, and identification is presented, which is supported by face identification from video data obtained from a steerable camera. Special attention is paid to latency aspects and online processing capabilities, as they are important for the application under investigation, namely ambient communication. It describes the vision of terminal-less, session-less and multi-modal telecommunication with remote partners, where the user can move freely within his home while the communication follows him. The speaker diarization serves as a context source, which has been integrated in a service-oriented middleware architecture and provided to the application to select the most appropriate I/O device and to steer the camera towards the speaker during ambient communication."}],"date_created":"2019-07-12T05:30:16Z","keyword":["audio streaming","audio visual data streaming","context information speech","face identification","face recognition","image segmentation","middleware","multimodal telecommunication","online diarization","service oriented middleware architecture","sessionless telecommunication","software architecture","speaker identification","speaker localization","speaker recognition","steerable camera","telecommunication computing","temporal segmentation","terminal-less telecommunication","video streaming"],"type":"journal_article","department":[{"_id":"54"}]},{"status":"public","user_id":"44006","volume":17,"page":"974-984","_id":"11937","citation":{"apa":"Windmann, S., &#38; Haeb-Umbach, R. (2009). Approaches to Iterative Speech Feature Enhancement and Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, <i>17</i>(5), 974–984. <a href=\"https://doi.org/10.1109/TASL.2009.2014894\">https://doi.org/10.1109/TASL.2009.2014894</a>","ieee":"S. Windmann and R. Haeb-Umbach, “Approaches to Iterative Speech Feature Enhancement and Recognition,” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 17, no. 5, pp. 974–984, 2009.","chicago":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Approaches to Iterative Speech Feature Enhancement and Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i> 17, no. 5 (2009): 974–84. <a href=\"https://doi.org/10.1109/TASL.2009.2014894\">https://doi.org/10.1109/TASL.2009.2014894</a>.","short":"S. Windmann, R. Haeb-Umbach, IEEE Transactions on Audio, Speech, and Language Processing 17 (2009) 974–984.","mla":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Approaches to Iterative Speech Feature Enhancement and Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 17, no. 5, 2009, pp. 974–84, doi:<a href=\"https://doi.org/10.1109/TASL.2009.2014894\">10.1109/TASL.2009.2014894</a>.","ama":"Windmann S, Haeb-Umbach R. Approaches to Iterative Speech Feature Enhancement and Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>. 2009;17(5):974-984. doi:<a href=\"https://doi.org/10.1109/TASL.2009.2014894\">10.1109/TASL.2009.2014894</a>","bibtex":"@article{Windmann_Haeb-Umbach_2009, title={Approaches to Iterative Speech Feature Enhancement and Recognition}, volume={17}, DOI={<a href=\"https://doi.org/10.1109/TASL.2009.2014894\">10.1109/TASL.2009.2014894</a>}, number={5}, journal={IEEE Transactions on Audio, Speech, and Language Processing}, author={Windmann, Stefan and Haeb-Umbach, Reinhold}, year={2009}, pages={974–984} }"},"oa":"1","date_updated":"2022-01-06T06:51:12Z","intvolume":"        17","title":"Approaches to Iterative Speech Feature Enhancement and Recognition","year":"2009","author":[{"last_name":"Windmann","first_name":"Stefan","full_name":"Windmann, Stefan"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"doi":"10.1109/TASL.2009.2014894","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2009/WiHa09-1.pdf"}],"language":[{"iso":"eng"}],"abstract":[{"text":"In automatic speech recognition, hidden Markov models (HMMs) are commonly used for speech decoding, while switching linear dynamic models (SLDMs) can be employed for a preceding model-based speech feature enhancement. In this paper, these model types are combined in order to obtain a novel iterative speech feature enhancement and recognition architecture. It is shown that speech feature enhancement with SLDMs can be improved by feeding back information from the HMM to the enhancement stage. Two different feedback structures are derived. In the first, the posteriors of the HMM states are used to control the model probabilities of the SLDMs, while in the second they are employed to directly influence the estimate of the speech feature distribution. Both approaches lead to improvements in recognition accuracy both on the AURORA2 and AURORA4 databases compared to non-iterative speech feature enhancement with SLDMs. It is also shown that a combination with uncertainty decoding further enhances performance.","lang":"eng"}],"issue":"5","publication":"IEEE Transactions on Audio, Speech, and Language Processing","type":"journal_article","keyword":["AURORA2 databases","AURORA4 databases","automatic speech recognition","feedback structures","hidden Markov models","HMM","iterative methods","iterative speech feature enhancement","model probabilities","speech decoding","speech enhancement","speech feature distribution","speech recognition","switching linear dynamic models"],"department":[{"_id":"54"}],"date_created":"2019-07-12T05:31:08Z"},{"oa":"1","citation":{"ieee":"S. Windmann and R. Haeb-Umbach, “Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition,” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 17, no. 8, pp. 1577–1590, 2009.","apa":"Windmann, S., &#38; Haeb-Umbach, R. (2009). Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, <i>17</i>(8), 1577–1590. <a href=\"https://doi.org/10.1109/TASL.2009.2023172\">https://doi.org/10.1109/TASL.2009.2023172</a>","short":"S. Windmann, R. Haeb-Umbach, IEEE Transactions on Audio, Speech, and Language Processing 17 (2009) 1577–1590.","chicago":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i> 17, no. 8 (2009): 1577–90. <a href=\"https://doi.org/10.1109/TASL.2009.2023172\">https://doi.org/10.1109/TASL.2009.2023172</a>.","mla":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 17, no. 8, 2009, pp. 1577–90, doi:<a href=\"https://doi.org/10.1109/TASL.2009.2023172\">10.1109/TASL.2009.2023172</a>.","bibtex":"@article{Windmann_Haeb-Umbach_2009, title={Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition}, volume={17}, DOI={<a href=\"https://doi.org/10.1109/TASL.2009.2023172\">10.1109/TASL.2009.2023172</a>}, number={8}, journal={IEEE Transactions on Audio, Speech, and Language Processing}, author={Windmann, Stefan and Haeb-Umbach, Reinhold}, year={2009}, pages={1577–1590} }","ama":"Windmann S, Haeb-Umbach R. Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>. 2009;17(8):1577-1590. doi:<a href=\"https://doi.org/10.1109/TASL.2009.2023172\">10.1109/TASL.2009.2023172</a>"},"user_id":"44006","volume":17,"page":"1577-1590","_id":"11938","status":"public","type":"journal_article","keyword":["AURORA4 database","blockwise EM algorithm","covariance analysis","linear state model","noise covariance","noise-robust automatic speech recognition","noisy speech cepstra","offline training mode","parameter estimation","speech recognition","speech recognition equipment","speech recognizer","state-space methods","state-space model"],"department":[{"_id":"54"}],"date_created":"2019-07-12T05:31:09Z","abstract":[{"lang":"eng","text":"In this paper, parameter estimation of a state-space model of noise or noisy speech cepstra is investigated. A blockwise EM algorithm is derived for the estimation of the state and observation noise covariance from noise-only input data. It is supposed to be used during the offline training mode of a speech recognizer. Further a sequential online EM algorithm is developed to adapt the observation noise covariance on noisy speech cepstra at its input. The estimated parameters are then used in model-based speech feature enhancement for noise-robust automatic speech recognition. Experiments on the AURORA4 database lead to improved recognition results with a linear state model compared to the assumption of stationary noise."}],"issue":"8","publication":"IEEE Transactions on Audio, Speech, and Language Processing","doi":"10.1109/TASL.2009.2023172","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2009/WiHa09-2.pdf","open_access":"1"}],"language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:12Z","intvolume":"        17","title":"Parameter Estimation of a State-Space Model of Noise for Robust Speech Recognition","year":"2009","author":[{"last_name":"Windmann","first_name":"Stefan","full_name":"Windmann, Stefan"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}]},{"doi":"10.1109/DEVLRN.2009.5175516","user_id":"14931","page":"1-6","_id":"17272","publisher":"IEEE","language":[{"iso":"eng"}],"date_updated":"2023-02-01T13:06:43Z","status":"public","title":"People modify their tutoring behavior in robot-directed interaction for action learning","year":"2009","author":[{"last_name":"Vollmer","first_name":"Anna-Lisa","full_name":"Vollmer, Anna-Lisa"},{"first_name":"Katrin Solveig","last_name":"Lohan","full_name":"Lohan, Katrin Solveig"},{"full_name":"Fischer, Kerstin","first_name":"Kerstin","last_name":"Fischer"},{"full_name":"Nagai, Yukie","first_name":"Yukie","last_name":"Nagai"},{"first_name":"Karola","last_name":"Pitsch","full_name":"Pitsch, Karola"},{"last_name":"Fritsch","first_name":"Jannik","full_name":"Fritsch, Jannik"},{"id":"50352","last_name":"Rohlfing","first_name":"Katharina","full_name":"Rohlfing, Katharina"},{"first_name":"Britta","last_name":"Wrede","full_name":"Wrede, Britta"}],"type":"conference","keyword":["robot simulation","hand movement velocity","robotic interaction partner","robotic agent","robot-directed interaction","multimodal analysis","Motionese","Motherese","intelligent tutoring systems","immature cognitive capability","human computer interaction","eye gaze","child-directed speech","child-directed motion","bottom-up system","bottom-up saliency-based attention model","adult-robot interaction","adult-child interaction","adult-adult interaction","human-robot interaction","action learning","social learning scenario","social robotics","software agents","top-down feedback structures","tutoring behavior"],"department":[{"_id":"749"}],"date_created":"2020-06-24T13:02:43Z","abstract":[{"text":"In developmental research, tutoring behavior has been identified as scaffolding infants' learning processes. It has been defined in terms of child-directed speech (Motherese), child-directed motion (Motionese), and contingency. In the field of developmental robotics, research often assumes that in human-robot interaction (HRI), robots are treated similar to infants, because their immature cognitive capabilities benefit from this behavior. However, according to our knowledge, it has barely been studied whether this is true and how exactly humans alter their behavior towards a robotic interaction partner. In this paper, we present results concerning the acceptance of a robotic agent in a social learning scenario obtained via comparison to adults and 8-11 months old infants in equal conditions. These results constitute an important empirical basis for making use of tutoring behavior in social robotics. In our study, we performed a detailed multimodal analysis of HRI in a tutoring situation using the example of a robot simulation equipped with a bottom-up saliency-based attention model. Our results reveal significant differences in hand movement velocity, motion pauses, range of motion, and eye gaze suggesting that for example adults decrease their hand movement velocity in an Adult-Child Interaction (ACI), opposed to an Adult-Adult Interaction (AAI) and this decrease is even higher in the Adult-Robot Interaction (ARI). We also found important differences between ACI and ARI in how the behavior is modified over time as the interaction unfolds. These findings indicate the necessity of integrating top-down feedback structures into a bottom-up system for robots to be fully accepted as interaction partners.","lang":"eng"}],"publication":"Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning","citation":{"mla":"Vollmer, Anna-Lisa, et al. “People Modify Their Tutoring Behavior in Robot-Directed Interaction for Action Learning.” <i>Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning</i>, IEEE, 2009, pp. 1–6, doi:<a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">10.1109/DEVLRN.2009.5175516</a>.","apa":"Vollmer, A.-L., Lohan, K. S., Fischer, K., Nagai, Y., Pitsch, K., Fritsch, J., Rohlfing, K., &#38; Wrede, B. (2009). People modify their tutoring behavior in robot-directed interaction for action learning. <i>Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning</i>, 1–6. <a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">https://doi.org/10.1109/DEVLRN.2009.5175516</a>","ieee":"A.-L. Vollmer <i>et al.</i>, “People modify their tutoring behavior in robot-directed interaction for action learning,” in <i>Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning</i>, 2009, pp. 1–6, doi: <a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">10.1109/DEVLRN.2009.5175516</a>.","short":"A.-L. Vollmer, K.S. Lohan, K. Fischer, Y. Nagai, K. Pitsch, J. Fritsch, K. Rohlfing, B. Wrede, in: Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning, IEEE, 2009, pp. 1–6.","ama":"Vollmer A-L, Lohan KS, Fischer K, et al. People modify their tutoring behavior in robot-directed interaction for action learning. In: <i>Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning</i>. IEEE; 2009:1-6. doi:<a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">10.1109/DEVLRN.2009.5175516</a>","chicago":"Vollmer, Anna-Lisa, Katrin Solveig Lohan, Kerstin Fischer, Yukie Nagai, Karola Pitsch, Jannik Fritsch, Katharina Rohlfing, and Britta Wrede. “People Modify Their Tutoring Behavior in Robot-Directed Interaction for Action Learning.” In <i>Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning</i>, 1–6. IEEE, 2009. <a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">https://doi.org/10.1109/DEVLRN.2009.5175516</a>.","bibtex":"@inproceedings{Vollmer_Lohan_Fischer_Nagai_Pitsch_Fritsch_Rohlfing_Wrede_2009, title={People modify their tutoring behavior in robot-directed interaction for action learning}, DOI={<a href=\"https://doi.org/10.1109/DEVLRN.2009.5175516\">10.1109/DEVLRN.2009.5175516</a>}, booktitle={Development and Learning, 2009. ICDL 2009. IEEE 8th International Conference on Development and Learning}, publisher={IEEE}, author={Vollmer, Anna-Lisa and Lohan, Katrin Solveig and Fischer, Kerstin and Nagai, Yukie and Pitsch, Karola and Fritsch, Jannik and Rohlfing, Katharina and Wrede, Britta}, year={2009}, pages={1–6} }"}},{"doi":"10.1109/TASL.2008.925879","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2008/IoHa08-1.pdf"}],"intvolume":"        16","date_updated":"2022-01-06T06:51:10Z","author":[{"last_name":"Ion","first_name":"Valentin","full_name":"Ion, Valentin"},{"id":"242","last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold"}],"year":"2008","title":"A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition","department":[{"_id":"54"}],"keyword":["automatic speech recognition","bit errors","codecs","communication links","corrupted observations","decoding","distributed speech recognition","error-prone communication network","feature vector sequence","hidden Markov model-based ASR","hidden Markov models","inter-frame correlation","Internet telephony","network speech recognition","packet loss","speech posterior","speech recognition","transmission error robust speech recognition","uncertainty decoding","voice-over-IP codecs"],"type":"journal_article","date_created":"2019-07-12T05:28:53Z","abstract":[{"text":"In this paper, we derive an uncertainty decoding rule for automatic speech recognition (ASR), which accounts for both corrupted observations and inter-frame correlation. The conditional independence assumption, prevalent in hidden Markov model-based ASR, is relaxed to obtain a clean speech posterior that is conditioned on the complete observed feature vector sequence. This is a more informative posterior than one conditioned only on the current observation. The novel decoding is used to obtain a transmission-error robust remote ASR system, where the speech capturing unit is connected to the decoder via an error-prone communication network. We show how the clean speech posterior can be computed for communication links being characterized by either bit errors or packet loss. Recognition results are presented for both distributed and network speech recognition, where in the latter case common voice-over-IP codecs are employed.","lang":"eng"}],"issue":"5","publication":"IEEE Transactions on Audio, Speech, and Language Processing","volume":16,"user_id":"44006","_id":"11820","page":"1047-1060","status":"public","oa":"1","citation":{"mla":"Ion, Valentin, and Reinhold Haeb-Umbach. “A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 16, no. 5, 2008, pp. 1047–60, doi:<a href=\"https://doi.org/10.1109/TASL.2008.925879\">10.1109/TASL.2008.925879</a>.","bibtex":"@article{Ion_Haeb-Umbach_2008, title={A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition}, volume={16}, DOI={<a href=\"https://doi.org/10.1109/TASL.2008.925879\">10.1109/TASL.2008.925879</a>}, number={5}, journal={IEEE Transactions on Audio, Speech, and Language Processing}, author={Ion, Valentin and Haeb-Umbach, Reinhold}, year={2008}, pages={1047–1060} }","ama":"Ion V, Haeb-Umbach R. A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>. 2008;16(5):1047-1060. doi:<a href=\"https://doi.org/10.1109/TASL.2008.925879\">10.1109/TASL.2008.925879</a>","ieee":"V. Ion and R. Haeb-Umbach, “A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition,” <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, vol. 16, no. 5, pp. 1047–1060, 2008.","apa":"Ion, V., &#38; Haeb-Umbach, R. (2008). A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition. <i>IEEE Transactions on Audio, Speech, and Language Processing</i>, <i>16</i>(5), 1047–1060. <a href=\"https://doi.org/10.1109/TASL.2008.925879\">https://doi.org/10.1109/TASL.2008.925879</a>","short":"V. Ion, R. Haeb-Umbach, IEEE Transactions on Audio, Speech, and Language Processing 16 (2008) 1047–1060.","chicago":"Ion, Valentin, and Reinhold Haeb-Umbach. “A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition.” <i>IEEE Transactions on Audio, Speech, and Language Processing</i> 16, no. 5 (2008): 1047–60. <a href=\"https://doi.org/10.1109/TASL.2008.925879\">https://doi.org/10.1109/TASL.2008.925879</a>."}},{"author":[{"full_name":"Warsitz, Ernst","last_name":"Warsitz","first_name":"Ernst"},{"first_name":"Alexander","last_name":"Krueger","full_name":"Krueger, Alexander"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"title":"Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller","year":"2008","status":"public","date_updated":"2022-01-06T06:51:12Z","_id":"11935","language":[{"iso":"eng"}],"page":"73-76","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2008/WaKrHa08.pdf","open_access":"1"}],"user_id":"44006","doi":"10.1109/ICASSP.2008.4517549","citation":{"ieee":"E. Warsitz, A. Krueger, and R. Haeb-Umbach, “Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 2008, pp. 73–76.","apa":"Warsitz, E., Krueger, A., &#38; Haeb-Umbach, R. (2008). Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i> (pp. 73–76). <a href=\"https://doi.org/10.1109/ICASSP.2008.4517549\">https://doi.org/10.1109/ICASSP.2008.4517549</a>","chicago":"Warsitz, Ernst, Alexander Krueger, and Reinhold Haeb-Umbach. “Speech Enhancement with a New Generalized Eigenvector Blocking Matrix for Application in a Generalized Sidelobe Canceller.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 73–76, 2008. <a href=\"https://doi.org/10.1109/ICASSP.2008.4517549\">https://doi.org/10.1109/ICASSP.2008.4517549</a>.","short":"E. Warsitz, A. Krueger, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008), 2008, pp. 73–76.","mla":"Warsitz, Ernst, et al. “Speech Enhancement with a New Generalized Eigenvector Blocking Matrix for Application in a Generalized Sidelobe Canceller.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 2008, pp. 73–76, doi:<a href=\"https://doi.org/10.1109/ICASSP.2008.4517549\">10.1109/ICASSP.2008.4517549</a>.","bibtex":"@inproceedings{Warsitz_Krueger_Haeb-Umbach_2008, title={Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2008.4517549\">10.1109/ICASSP.2008.4517549</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}, author={Warsitz, Ernst and Krueger, Alexander and Haeb-Umbach, Reinhold}, year={2008}, pages={73–76} }","ama":"Warsitz E, Krueger A, Haeb-Umbach R. Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>. ; 2008:73-76. doi:<a href=\"https://doi.org/10.1109/ICASSP.2008.4517549\">10.1109/ICASSP.2008.4517549</a>"},"publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)","abstract":[{"text":"The generalized sidelobe canceller by Griffith and Jim is a robust beamforming method to enhance a desired (speech) signal in the presence of stationary noise. Its performance depends to a high degree on the construction of the blocking matrix which produces noise reference signals for the subsequent adaptive interference canceller. Especially in reverberated environments the beamformer may suffer from signal leakage and reduced noise suppression. In this paper a new blocking matrix is proposed. It is based on a generalized eigenvalue problem whose solution provides an indirect estimation of the transfer functions from the source to the sensors. The quality of the new generalized eigenvector blocking matrix is studied in simulated rooms with different reverberation times and is compared to alternatives proposed in the literature.","lang":"eng"}],"date_created":"2019-07-12T05:31:06Z","department":[{"_id":"54"}],"oa":"1","type":"conference","keyword":["adaptive interference canceller","adaptive signal processing","array signal processing","beamforming method","eigenvalues and eigenfunctions","generalized eigenvector blocking matrix","generalized sidelobe canceller","interference suppression","matrix algebra","noise suppression","speech enhancement","transfer function estimation","transfer functions"]},{"oa":"1","department":[{"_id":"54"}],"type":"conference","keyword":["a posteriori probability","AURORA2 database","Bayesian inference","Bayes methods","channel bank filters","extended Kalman filter banks","hidden noise state variable","Kalman filters","noise dynamics","speech enhancement","speech feature enhancement","speech feature trajectory","switching linear dynamical model approach"],"date_created":"2019-07-12T05:31:11Z","abstract":[{"lang":"eng","text":"In this paper a switching linear dynamical model (SLDM) approach for speech feature enhancement is improved by employing more accurate models for the dynamics of speech and noise. The model of the clean speech feature trajectory is improved by augmenting the state vector to capture information derived from the delta features. Further a hidden noise state variable is introduced to obtain a more elaborated model for the noise dynamics. Approximate Bayesian inference in the SLDM is carried out by a bank of extended Kalman filters, whose outputs are combined according to the a posteriori probability of the individual state models. Experimental results on the AURORA2 database show improved recognition accuracy."}],"citation":{"bibtex":"@inproceedings{Windmann_Haeb-Umbach_2008, title={Modeling the dynamics of speech and noise for speech feature enhancement in ASR}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2008.4518633\">10.1109/ICASSP.2008.4518633</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}, author={Windmann, Stefan and Haeb-Umbach, Reinhold}, year={2008}, pages={4409–4412} }","ama":"Windmann S, Haeb-Umbach R. Modeling the dynamics of speech and noise for speech feature enhancement in ASR. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>. ; 2008:4409-4412. doi:<a href=\"https://doi.org/10.1109/ICASSP.2008.4518633\">10.1109/ICASSP.2008.4518633</a>","mla":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Modeling the Dynamics of Speech and Noise for Speech Feature Enhancement in ASR.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 2008, pp. 4409–12, doi:<a href=\"https://doi.org/10.1109/ICASSP.2008.4518633\">10.1109/ICASSP.2008.4518633</a>.","short":"S. Windmann, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008), 2008, pp. 4409–4412.","chicago":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Modeling the Dynamics of Speech and Noise for Speech Feature Enhancement in ASR.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 4409–12, 2008. <a href=\"https://doi.org/10.1109/ICASSP.2008.4518633\">https://doi.org/10.1109/ICASSP.2008.4518633</a>.","ieee":"S. Windmann and R. Haeb-Umbach, “Modeling the dynamics of speech and noise for speech feature enhancement in ASR,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i>, 2008, pp. 4409–4412.","apa":"Windmann, S., &#38; Haeb-Umbach, R. (2008). Modeling the dynamics of speech and noise for speech feature enhancement in ASR. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)</i> (pp. 4409–4412). <a href=\"https://doi.org/10.1109/ICASSP.2008.4518633\">https://doi.org/10.1109/ICASSP.2008.4518633</a>"},"publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)","doi":"10.1109/ICASSP.2008.4518633","user_id":"44006","_id":"11939","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2008/WiHa08-1.pdf"}],"page":"4409-4412","date_updated":"2022-01-06T06:51:12Z","author":[{"first_name":"Stefan","last_name":"Windmann","full_name":"Windmann, Stefan"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"title":"Modeling the dynamics of speech and noise for speech feature enhancement in ASR","status":"public","year":"2008"},{"type":"conference","keyword":["discursive behavior","autonomous robot","BIRON","man-machine systems","robot abilities","robot knowledge","user gestures","robot verbal feedback utterance","speech processing","user verbal behavior","service robots","human-robot interaction","human computer interaction","gesture recognition"],"department":[{"_id":"749"}],"date_created":"2020-06-24T13:02:49Z","abstract":[{"text":"This paper investigates the influence of feedback provided by an autonomous robot (BIRON) on users’ discursive behavior. A user study is described during which users show objects to the robot. The results of the experiment indicate, that the robot’s verbal feedback utterances cause the humans to adapt their own way of speaking. The changes in users’ verbal behavior are due to their beliefs about the robots knowledge and abilities. In this paper they are identified and grouped. Moreover, the data implies variations in user behavior regarding gestures. Unlike speech, the robot was not able to give feedback with gestures. Due to the lack of feedback, users did not seem to have a consistent mental representation of the robot’s abilities to recognize gestures. As a result, changes between different gestures are interpreted to be unconscious variations accompanying speech.","lang":"eng"}],"citation":{"chicago":"Lohse, Manja, Katharina Rohlfing, Britta Wrede, and Gerhard Sagerer. “‘Try Something Else!’ — When Users Change Their Discursive Behavior in Human-Robot Interaction,” 3481–86, 2008. <a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">https://doi.org/10.1109/ROBOT.2008.4543743</a>.","short":"M. Lohse, K. Rohlfing, B. Wrede, G. Sagerer, in: 2008, pp. 3481–3486.","apa":"Lohse, M., Rohlfing, K., Wrede, B., &#38; Sagerer, G. (2008). <i>“Try something else!” — When users change their discursive behavior in human-robot interaction</i>. 3481–3486. <a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">https://doi.org/10.1109/ROBOT.2008.4543743</a>","ieee":"M. Lohse, K. Rohlfing, B. Wrede, and G. Sagerer, “‘Try something else!’ — When users change their discursive behavior in human-robot interaction,” 2008, pp. 3481–3486, doi: <a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">10.1109/ROBOT.2008.4543743</a>.","ama":"Lohse M, Rohlfing K, Wrede B, Sagerer G. “Try something else!” — When users change their discursive behavior in human-robot interaction. In: ; 2008:3481-3486. doi:<a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">10.1109/ROBOT.2008.4543743</a>","bibtex":"@inproceedings{Lohse_Rohlfing_Wrede_Sagerer_2008, title={“Try something else!” — When users change their discursive behavior in human-robot interaction}, DOI={<a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">10.1109/ROBOT.2008.4543743</a>}, author={Lohse, Manja and Rohlfing, Katharina and Wrede, Britta and Sagerer, Gerhard}, year={2008}, pages={3481–3486} }","mla":"Lohse, Manja, et al. <i>“Try Something Else!” — When Users Change Their Discursive Behavior in Human-Robot Interaction</i>. 2008, pp. 3481–86, doi:<a href=\"https://doi.org/10.1109/ROBOT.2008.4543743\">10.1109/ROBOT.2008.4543743</a>."},"doi":"10.1109/ROBOT.2008.4543743","user_id":"14931","page":"3481-3486","_id":"17278","language":[{"iso":"eng"}],"date_updated":"2023-02-01T13:08:20Z","title":"“Try something else!” — When users change their discursive behavior in human-robot interaction","year":"2008","status":"public","publication_identifier":{"isbn":["1050-4729"]},"author":[{"last_name":"Lohse","first_name":"Manja","full_name":"Lohse, Manja"},{"first_name":"Katharina","last_name":"Rohlfing","full_name":"Rohlfing, Katharina","id":"50352"},{"last_name":"Wrede","first_name":"Britta","full_name":"Wrede, Britta"},{"last_name":"Sagerer","first_name":"Gerhard","full_name":"Sagerer, Gerhard"}]},{"user_id":"44006","volume":1,"page":"I","_id":"11824","status":"public","oa":"1","citation":{"mla":"Ion, Valentin, and Reinhold Haeb-Umbach. “An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, vol. 1, 2006, p. I, doi:<a href=\"https://doi.org/10.1109/ICASSP.2006.1659984\">10.1109/ICASSP.2006.1659984</a>.","ama":"Ion V, Haeb-Umbach R. An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>. Vol 1. ; 2006:I. doi:<a href=\"https://doi.org/10.1109/ICASSP.2006.1659984\">10.1109/ICASSP.2006.1659984</a>","bibtex":"@inproceedings{Ion_Haeb-Umbach_2006, title={An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features}, volume={1}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2006.1659984\">10.1109/ICASSP.2006.1659984</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)}, author={Ion, Valentin and Haeb-Umbach, Reinhold}, year={2006}, pages={I} }","apa":"Ion, V., &#38; Haeb-Umbach, R. (2006). An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i> (Vol. 1, p. I). <a href=\"https://doi.org/10.1109/ICASSP.2006.1659984\">https://doi.org/10.1109/ICASSP.2006.1659984</a>","ieee":"V. Ion and R. Haeb-Umbach, “An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, 2006, vol. 1, p. I.","short":"V. Ion, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006), 2006, p. I.","chicago":"Ion, Valentin, and Reinhold Haeb-Umbach. “An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, 1:I, 2006. <a href=\"https://doi.org/10.1109/ICASSP.2006.1659984\">https://doi.org/10.1109/ICASSP.2006.1659984</a>."},"doi":"10.1109/ICASSP.2006.1659984","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2006/IoHa06-2.pdf","open_access":"1"}],"language":[{"iso":"eng"}],"date_updated":"2022-01-06T06:51:10Z","intvolume":"         1","year":"2006","title":"An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features","author":[{"first_name":"Valentin","last_name":"Ion","full_name":"Ion, Valentin"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"type":"conference","keyword":["distributed speech recognition","least mean squares methods","MAP estimate","maximum likelihood estimation","MMSE estimate","packet loss compensation scheme","packet switched communication","posteriori probability density function","robust error mitigation method","soft-features","speech recognition","table lookup","voice communication","wireless channels"],"department":[{"_id":"54"}],"date_created":"2019-07-12T05:28:58Z","abstract":[{"text":"Soft-feature based speech recognition, which is an example of uncertainty decoding, has been proven to be a robust error mitigation method for distributed speech recognition over wireless channels exhibiting bit errors. In this paper we extend this concept to packet-oriented transmissions. The a posteriori probability density function of the lost feature vector, given the closest received neighbours, is computed. In the experiments, the nearest frame repetition, which is shown to be equivalent to the MAP estimate, outperforms the MMSE estimate for long bursts. Taking the variance into account at the speech recognition stage results in superior performance compared to classical schemes using point estimates. A computationally and memory efficient implementation of the proposed packet loss compensation scheme based on table lookup is presented","lang":"eng"}],"publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)"},{"date_created":"2019-07-12T05:28:59Z","keyword":["Channel error robustness","Distributed speech recognition","Soft features","Uncertainty decoding"],"type":"journal_article","department":[{"_id":"54"}],"publication":"Speech Communication","issue":"11","abstract":[{"lang":"eng","text":"In this paper, we propose an enhanced error concealment strategy at the server side of a distributed speech recognition (DSR) system, which is fully compatible with the existing DSR standard. It is based on a Bayesian approach, where the a posteriori probability density of the error-free feature vector is computed, given all received feature vectors which are possibly corrupted by transmission errors. Rather than computing a point estimate, such as the MMSE estimate, and plugging it into the Bayesian decision rule, we employ uncertainty decoding, which results in an integration over the uncertainty in the feature domain. In a typical scenario the communication between the thin client, often a mobile device, and the recognition server spreads across heterogeneous networks. Both bit errors on circuit-switched links and lost data packets on IP connections are mitigated by our approach in a unified manner. The experiments reveal improved robustness both for small- and large-vocabulary recognition tasks."}],"main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2006/IoHa06-3.pdf","open_access":"1"}],"language":[{"iso":"eng"}],"doi":"10.1016/j.specom.2006.03.007","title":"Uncertainty decoding for distributed speech recognition over error-prone networks","year":"2006","author":[{"full_name":"Ion, Valentin","last_name":"Ion","first_name":"Valentin"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"date_updated":"2022-01-06T06:51:10Z","intvolume":"        48","oa":"1","citation":{"mla":"Ion, Valentin, and Reinhold Haeb-Umbach. “Uncertainty Decoding for Distributed Speech Recognition over Error-Prone Networks.” <i>Speech Communication</i>, vol. 48, no. 11, 2006, pp. 1435–46, doi:<a href=\"https://doi.org/10.1016/j.specom.2006.03.007\">10.1016/j.specom.2006.03.007</a>.","bibtex":"@article{Ion_Haeb-Umbach_2006, title={Uncertainty decoding for distributed speech recognition over error-prone networks}, volume={48}, DOI={<a href=\"https://doi.org/10.1016/j.specom.2006.03.007\">10.1016/j.specom.2006.03.007</a>}, number={11}, journal={Speech Communication}, author={Ion, Valentin and Haeb-Umbach, Reinhold}, year={2006}, pages={1435–1446} }","ama":"Ion V, Haeb-Umbach R. Uncertainty decoding for distributed speech recognition over error-prone networks. <i>Speech Communication</i>. 2006;48(11):1435-1446. doi:<a href=\"https://doi.org/10.1016/j.specom.2006.03.007\">10.1016/j.specom.2006.03.007</a>","ieee":"V. Ion and R. Haeb-Umbach, “Uncertainty decoding for distributed speech recognition over error-prone networks,” <i>Speech Communication</i>, vol. 48, no. 11, pp. 1435–1446, 2006.","apa":"Ion, V., &#38; Haeb-Umbach, R. (2006). Uncertainty decoding for distributed speech recognition over error-prone networks. <i>Speech Communication</i>, <i>48</i>(11), 1435–1446. <a href=\"https://doi.org/10.1016/j.specom.2006.03.007\">https://doi.org/10.1016/j.specom.2006.03.007</a>","short":"V. Ion, R. Haeb-Umbach, Speech Communication 48 (2006) 1435–1446.","chicago":"Ion, Valentin, and Reinhold Haeb-Umbach. “Uncertainty Decoding for Distributed Speech Recognition over Error-Prone Networks.” <i>Speech Communication</i> 48, no. 11 (2006): 1435–46. <a href=\"https://doi.org/10.1016/j.specom.2006.03.007\">https://doi.org/10.1016/j.specom.2006.03.007</a>."},"page":"1435-1446","_id":"11825","user_id":"44006","volume":48,"status":"public"},{"date_created":"2019-07-12T05:31:15Z","keyword":["clean speech training data","iterative methods","iterative speech enhancement","Kalman filter","Kalman filters","Kalman-LM-iterative algorithm","line spectral pair parameters","log-spectral distance","marginalized particle filter","noise level","nonlinear dynamic state speech model","particle filtering (numerical methods)","single channel speech enhancement","SNR gains","speech enhancement","speech samples"],"type":"conference","department":[{"_id":"54"}],"publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)","abstract":[{"text":"A marginalized particle filter is proposed for performing single channel speech enhancement with a non-linear dynamic state model. The system consists of a particle filter for tracking line spectral pair (LSP) parameters and a Kalman filter per particle for speech enhancement. The state model for the LSPs has been learnt on clean speech training data. In our approach parameters and speech samples are processed at different time scales by assuming the parameters to be constant for small blocks of data. Further enhancement is obtained by an iteration which can be applied on these small blocks. The experiments show that similar SNR gains are obtained as with the Kalman-LM-iterative algorithm. However better values of the noise level and the log-spectral distance are achieved","lang":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2006/WiHa06-2.pdf"}],"language":[{"iso":"eng"}],"doi":"10.1109/ICASSP.2006.1660058","year":"2006","title":"Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters","author":[{"first_name":"Stefan","last_name":"Windmann","full_name":"Windmann, Stefan"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"date_updated":"2022-01-06T06:51:12Z","intvolume":"         1","oa":"1","citation":{"ieee":"S. Windmann and R. Haeb-Umbach, “Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, 2006, vol. 1, p. I.","apa":"Windmann, S., &#38; Haeb-Umbach, R. (2006). Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i> (Vol. 1, p. I). <a href=\"https://doi.org/10.1109/ICASSP.2006.1660058\">https://doi.org/10.1109/ICASSP.2006.1660058</a>","short":"S. Windmann, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006), 2006, p. I.","chicago":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Iterative Speech Enhancement Using a Non-Linear Dynamic State Model of Speech and Its Parameters.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, 1:I, 2006. <a href=\"https://doi.org/10.1109/ICASSP.2006.1660058\">https://doi.org/10.1109/ICASSP.2006.1660058</a>.","mla":"Windmann, Stefan, and Reinhold Haeb-Umbach. “Iterative Speech Enhancement Using a Non-Linear Dynamic State Model of Speech and Its Parameters.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>, vol. 1, 2006, p. I, doi:<a href=\"https://doi.org/10.1109/ICASSP.2006.1660058\">10.1109/ICASSP.2006.1660058</a>.","bibtex":"@inproceedings{Windmann_Haeb-Umbach_2006, title={Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters}, volume={1}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2006.1660058\">10.1109/ICASSP.2006.1660058</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)}, author={Windmann, Stefan and Haeb-Umbach, Reinhold}, year={2006}, pages={I} }","ama":"Windmann S, Haeb-Umbach R. Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)</i>. Vol 1. ; 2006:I. doi:<a href=\"https://doi.org/10.1109/ICASSP.2006.1660058\">10.1109/ICASSP.2006.1660058</a>"},"page":"I","_id":"11943","user_id":"44006","volume":1,"status":"public"},{"citation":{"mla":"Ion, Valentin, and Reinhold Haeb-Umbach. “A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services.” <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)</i>, vol. 1, 2005, pp. 333–36, doi:<a href=\"https://doi.org/10.1109/ICASSP.2005.1415118\">10.1109/ICASSP.2005.1415118</a>.","ama":"Ion V, Haeb-Umbach R. A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services. In: <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)</i>. Vol 1. ; 2005:333-336. doi:<a href=\"https://doi.org/10.1109/ICASSP.2005.1415118\">10.1109/ICASSP.2005.1415118</a>","bibtex":"@inproceedings{Ion_Haeb-Umbach_2005, title={A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services}, volume={1}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2005.1415118\">10.1109/ICASSP.2005.1415118</a>}, booktitle={IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)}, author={Ion, Valentin and Haeb-Umbach, Reinhold}, year={2005}, pages={333–336} }","apa":"Ion, V., &#38; Haeb-Umbach, R. (2005). A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services. In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)</i> (Vol. 1, pp. 333–336). <a href=\"https://doi.org/10.1109/ICASSP.2005.1415118\">https://doi.org/10.1109/ICASSP.2005.1415118</a>","ieee":"V. Ion and R. Haeb-Umbach, “A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services,” in <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)</i>, 2005, vol. 1, pp. 333–336.","short":"V. Ion, R. Haeb-Umbach, in: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005), 2005, pp. 333–336.","chicago":"Ion, Valentin, and Reinhold Haeb-Umbach. “A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services.” In <i>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)</i>, 1:333–36, 2005. <a href=\"https://doi.org/10.1109/ICASSP.2005.1415118\">https://doi.org/10.1109/ICASSP.2005.1415118</a>."},"oa":"1","status":"public","_id":"11828","page":"333-336","volume":1,"user_id":"44006","publication":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)","abstract":[{"text":"In this paper we present a comparison of the recently proposed Soft-Feature Distributed Speech Recognition (SFDSR) with the two evaluated candidate codecs for Speech Enabled Services over wireless networks: Adaptive Multirate Codec (AMR) and the ETSI Extended Advanced Front-End for Distributed Speech Recognition (XAFE). It is shown that SFDSR achieves the best recognition performance on a simulated GSM transmission, followed by XAFE and AMR.We also present some new results concerning SFDSR which demonstrate the versatility of the approach. Further, a simple method is introduced which considerably reduces the computational effort.","lang":"eng"}],"date_created":"2019-07-12T05:29:02Z","department":[{"_id":"54"}],"type":"conference","keyword":["adaptive codes","adaptive multirate codec","AMR","distributed speech recognition","ETSI","extended advanced front-end","recognition performance","SFDSR","simulated GSM transmission","soft-feature distributed speech recognition","speech codecs","speech coding","speech recognition","variable rate codes","XAFE"],"author":[{"last_name":"Ion","first_name":"Valentin","full_name":"Ion, Valentin"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"title":"A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services","year":"2005","intvolume":"         1","date_updated":"2022-01-06T06:51:10Z","language":[{"iso":"eng"}],"main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2005/IoHa05-2.pdf"}],"doi":"10.1109/ICASSP.2005.1415118"},{"doi":"10.1109/MMSP.2004.1436569","user_id":"44006","language":[{"iso":"eng"}],"_id":"11931","page":"367-370","main_file_link":[{"url":"https://groups.uni-paderborn.de/nt/pubs/2004/WaHa04.pdf","open_access":"1"}],"date_updated":"2022-01-06T06:51:12Z","author":[{"full_name":"Warsitz, Ernst","first_name":"Ernst","last_name":"Warsitz"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"year":"2004","status":"public","title":"Robust speaker direction estimation with particle filtering","oa":"1","department":[{"_id":"54"}],"type":"conference","keyword":["bimodal human-robot interface","binaural signal processing","enhanced single-channel input signal","filter-and-sum beamforming","filtering theory","FIR filter coefficient","generalized cross correlation method","microphones","microphone signal","nonlinear Bayesian tracking","particle filtering","robust adaptive algorithm","robust speaker direction estimation","signal processing","speech enhancement","speech recognition","speech recognizer","user interfaces"],"date_created":"2019-07-12T05:31:01Z","abstract":[{"text":"The paper is concerned with binaural signal processing for a bimodal human-robot interface with hearing and vision. The two microphone signals are processed to obtain an enhanced single-channel input signal for the subsequent speech recognizer and to localize the acoustic source, an important information for establishing a natural human-robot communication. We utilize a robust adaptive algorithm for filter-and-sum beamforming (FSB) and extract speaker direction information from the resulting FIR filter coefficients. Further, particle filtering is applied which conducts a nonlinear Bayesian tracking of speaker movement. Good location accuracy can be achieved even in highly reverberant environments. The results obtained outperform the conventional generalized cross correlation (GCC) method.","lang":"eng"}],"citation":{"bibtex":"@inproceedings{Warsitz_Haeb-Umbach_2004, title={Robust speaker direction estimation with particle filtering}, DOI={<a href=\"https://doi.org/10.1109/MMSP.2004.1436569\">10.1109/MMSP.2004.1436569</a>}, booktitle={IEEE Workshop on Multimedia Signal Processing (MMSP 2004)}, author={Warsitz, Ernst and Haeb-Umbach, Reinhold}, year={2004}, pages={367–370} }","ama":"Warsitz E, Haeb-Umbach R. Robust speaker direction estimation with particle filtering. In: <i>IEEE Workshop on Multimedia Signal Processing (MMSP 2004)</i>. ; 2004:367-370. doi:<a href=\"https://doi.org/10.1109/MMSP.2004.1436569\">10.1109/MMSP.2004.1436569</a>","mla":"Warsitz, Ernst, and Reinhold Haeb-Umbach. “Robust Speaker Direction Estimation with Particle Filtering.” <i>IEEE Workshop on Multimedia Signal Processing (MMSP 2004)</i>, 2004, pp. 367–70, doi:<a href=\"https://doi.org/10.1109/MMSP.2004.1436569\">10.1109/MMSP.2004.1436569</a>.","short":"E. Warsitz, R. Haeb-Umbach, in: IEEE Workshop on Multimedia Signal Processing (MMSP 2004), 2004, pp. 367–370.","chicago":"Warsitz, Ernst, and Reinhold Haeb-Umbach. “Robust Speaker Direction Estimation with Particle Filtering.” In <i>IEEE Workshop on Multimedia Signal Processing (MMSP 2004)</i>, 367–70, 2004. <a href=\"https://doi.org/10.1109/MMSP.2004.1436569\">https://doi.org/10.1109/MMSP.2004.1436569</a>.","ieee":"E. Warsitz and R. Haeb-Umbach, “Robust speaker direction estimation with particle filtering,” in <i>IEEE Workshop on Multimedia Signal Processing (MMSP 2004)</i>, 2004, pp. 367–370.","apa":"Warsitz, E., &#38; Haeb-Umbach, R. (2004). Robust speaker direction estimation with particle filtering. In <i>IEEE Workshop on Multimedia Signal Processing (MMSP 2004)</i> (pp. 367–370). <a href=\"https://doi.org/10.1109/MMSP.2004.1436569\">https://doi.org/10.1109/MMSP.2004.1436569</a>"},"publication":"IEEE Workshop on Multimedia Signal Processing (MMSP 2004)"},{"abstract":[{"text":"Portable devices come with different limitations in user interaction like limited display size, small keyboard, and different sorts of input and output capabilities. With the advance of speech recognition and speech synthesis technologies, their complementary use becomes attractive for mobile devices in order to implement real multimodal user interaction. However, current systems and formats do not sufficiently integrate advanced multimodal interactions. We introduce an advanced generic multimodal interaction and rendering system (MIRS) dedicated for mobile devices. MIRS incorporates efficient processing of XML specification languages for limited, mobile devices and comes with the XML-based dialog and interface specification language (DISL). DISL can be considered as an UIML subset, which is enhanced by the means of state-oriented dialog specifications. The dialog specification is based on ODSN (object oriented dialog specification notation), which has been introduced to define user interface control by means of interaction states with transition rules.","lang":"eng"}],"citation":{"apa":"Müller, W., Schäfer, R., &#38; Bleul, S. (2004). Interactive Multimodal User Interfaces for Mobile Devices. <i>Proceedings of HICCS-37</i>. 37th Annual Hawaii International Conference on System Sciences, Waikoloa, HI, USA. <a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">https://doi.org/10.1109/HICSS.2004.1265674</a>","ieee":"W. Müller, R. Schäfer, and S. Bleul, “Interactive Multimodal User Interfaces for Mobile Devices,” presented at the 37th Annual Hawaii International Conference on System Sciences, Waikoloa, HI, USA, 2004, doi: <a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">10.1109/HICSS.2004.1265674</a>.","short":"W. Müller, R. Schäfer, S. Bleul, in: Proceedings of HICCS-37, Waikoloa, HI, USA, 2004.","chicago":"Müller, Wolfgang, Robbie Schäfer, and Steffen Bleul. “Interactive Multimodal User Interfaces for Mobile Devices.” In <i>Proceedings of HICCS-37</i>. Waikoloa, HI, USA, 2004. <a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">https://doi.org/10.1109/HICSS.2004.1265674</a>.","mla":"Müller, Wolfgang, et al. “Interactive Multimodal User Interfaces for Mobile Devices.” <i>Proceedings of HICCS-37</i>, 2004, doi:<a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">10.1109/HICSS.2004.1265674</a>.","ama":"Müller W, Schäfer R, Bleul S. Interactive Multimodal User Interfaces for Mobile Devices. In: <i>Proceedings of HICCS-37</i>. ; 2004. doi:<a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">10.1109/HICSS.2004.1265674</a>","bibtex":"@inproceedings{Müller_Schäfer_Bleul_2004, place={Waikoloa, HI, USA}, title={Interactive Multimodal User Interfaces for Mobile Devices}, DOI={<a href=\"https://doi.org/10.1109/HICSS.2004.1265674\">10.1109/HICSS.2004.1265674</a>}, booktitle={Proceedings of HICCS-37}, author={Müller, Wolfgang and Schäfer, Robbie and Bleul, Steffen}, year={2004} }"},"publication":"Proceedings of HICCS-37","department":[{"_id":"672"}],"keyword":["User interfaces","Speech recognition","Streaming media","Specification languages","Keyboards","Speech synthesis","Rendering (computer graphics)","Ambient intelligence","Humans","Displays"],"type":"conference","date_created":"2023-01-24T08:46:31Z","place":"Waikoloa, HI, USA","date_updated":"2023-01-24T08:46:37Z","publication_identifier":{"isbn":["0-7695-2056-1"]},"author":[{"full_name":"Müller, Wolfgang","first_name":"Wolfgang","last_name":"Müller","id":"16243"},{"full_name":"Schäfer, Robbie","last_name":"Schäfer","first_name":"Robbie"},{"full_name":"Bleul, Steffen","last_name":"Bleul","first_name":"Steffen"}],"conference":{"name":"37th Annual Hawaii International Conference on System Sciences","location":"Waikoloa, HI, USA"},"title":"Interactive Multimodal User Interfaces for Mobile Devices","status":"public","year":"2004","user_id":"5786","doi":"10.1109/HICSS.2004.1265674","_id":"39053","language":[{"iso":"eng"}]},{"issue":"3","publication":"IEEE Transactions on Speech and Audio Processing","abstract":[{"text":"In this paper, it is shown that a correlation criterion is the appropriate criterion for bottom-up clustering to obtain broad phonetic class regression trees for maximum likelihood linear regression (MLLR)-based speaker adaptation. The correlation structure among speech units is estimated on the speaker-independent training data. In adaptation experiments the tree outperformed a regression tree obtained from clustering according to closeness in acoustic space and achieved results comparable with those of a manually designed broad phonetic class tree","lang":"eng"}],"date_created":"2019-07-12T05:28:04Z","type":"journal_article","keyword":["acoustic space","adaptation experiments","automatic generation","bottom-up clustering","broad phonetic class regression trees","correlation criterion","correlation methods","maximum likelihood estimation","maximum likelihood linear regression based speaker adaptation","MLLR adaptation","pattern clustering","phonetic regression class trees","speaker-independent training data","speech recognition","speech units","statistical analysis","trees (mathematics)"],"department":[{"_id":"54"}],"year":"2001","title":"Automatic generation of phonetic regression class trees for MLLR adaptation","author":[{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"date_updated":"2022-01-06T06:51:08Z","intvolume":"         9","main_file_link":[{"open_access":"1","url":"https://groups.uni-paderborn.de/nt/pubs/2001/Ha01.pdf"}],"language":[{"iso":"eng"}],"doi":"10.1109/89.906003","citation":{"apa":"Haeb-Umbach, R. (2001). Automatic generation of phonetic regression class trees for MLLR adaptation. <i>IEEE Transactions on Speech and Audio Processing</i>, <i>9</i>(3), 299–302. <a href=\"https://doi.org/10.1109/89.906003\">https://doi.org/10.1109/89.906003</a>","ieee":"R. Haeb-Umbach, “Automatic generation of phonetic regression class trees for MLLR adaptation,” <i>IEEE Transactions on Speech and Audio Processing</i>, vol. 9, no. 3, pp. 299–302, 2001.","short":"R. Haeb-Umbach, IEEE Transactions on Speech and Audio Processing 9 (2001) 299–302.","chicago":"Haeb-Umbach, Reinhold. “Automatic Generation of Phonetic Regression Class Trees for MLLR Adaptation.” <i>IEEE Transactions on Speech and Audio Processing</i> 9, no. 3 (2001): 299–302. <a href=\"https://doi.org/10.1109/89.906003\">https://doi.org/10.1109/89.906003</a>.","mla":"Haeb-Umbach, Reinhold. “Automatic Generation of Phonetic Regression Class Trees for MLLR Adaptation.” <i>IEEE Transactions on Speech and Audio Processing</i>, vol. 9, no. 3, 2001, pp. 299–302, doi:<a href=\"https://doi.org/10.1109/89.906003\">10.1109/89.906003</a>.","ama":"Haeb-Umbach R. Automatic generation of phonetic regression class trees for MLLR adaptation. <i>IEEE Transactions on Speech and Audio Processing</i>. 2001;9(3):299-302. doi:<a href=\"https://doi.org/10.1109/89.906003\">10.1109/89.906003</a>","bibtex":"@article{Haeb-Umbach_2001, title={Automatic generation of phonetic regression class trees for MLLR adaptation}, volume={9}, DOI={<a href=\"https://doi.org/10.1109/89.906003\">10.1109/89.906003</a>}, number={3}, journal={IEEE Transactions on Speech and Audio Processing}, author={Haeb-Umbach, Reinhold}, year={2001}, pages={299–302} }"},"oa":"1","status":"public","page":"299-302","_id":"11778","user_id":"44006","volume":9},{"title":"Hardware/Software Codesign in Speech Compression Applications","status":"public","year":"2000","author":[{"id":"16153","first_name":"Christian","orcid":"0000-0001-5728-9982","last_name":"Plessl","full_name":"Plessl, Christian"},{"full_name":"Maurer, Simon","last_name":"Maurer","first_name":"Simon"}],"date_updated":"2022-01-06T06:56:17Z","_id":"2433","publisher":"Computer Engineering and Networks Lab, ETH Zurich, Switzerland","user_id":"24135","citation":{"bibtex":"@book{Plessl_Maurer_2000, title={Hardware/Software Codesign in Speech Compression Applications}, publisher={Computer Engineering and Networks Lab, ETH Zurich, Switzerland}, author={Plessl, Christian and Maurer, Simon}, year={2000} }","ama":"Plessl C, Maurer S. <i>Hardware/Software Codesign in Speech Compression Applications</i>. Computer Engineering and Networks Lab, ETH Zurich, Switzerland; 2000.","mla":"Plessl, Christian, and Simon Maurer. <i>Hardware/Software Codesign in Speech Compression Applications</i>. Computer Engineering and Networks Lab, ETH Zurich, Switzerland, 2000.","short":"C. Plessl, S. Maurer, Hardware/Software Codesign in Speech Compression Applications, Computer Engineering and Networks Lab, ETH Zurich, Switzerland, 2000.","chicago":"Plessl, Christian, and Simon Maurer. <i>Hardware/Software Codesign in Speech Compression Applications</i>. Computer Engineering and Networks Lab, ETH Zurich, Switzerland, 2000.","ieee":"C. Plessl and S. Maurer, <i>Hardware/Software Codesign in Speech Compression Applications</i>. Computer Engineering and Networks Lab, ETH Zurich, Switzerland, 2000.","apa":"Plessl, C., &#38; Maurer, S. (2000). <i>Hardware/Software Codesign in Speech Compression Applications</i>. Computer Engineering and Networks Lab, ETH Zurich, Switzerland."},"date_created":"2018-04-17T15:56:00Z","keyword":["co-design","speech processing"],"type":"mastersthesis","department":[{"_id":"518"}]}]
