@inproceedings{11849,
  abstract     = {{In this contribution we investigate the effectiveness of Bayesian feature enhancement (BFE) on a medium-sized recognition task containing real-world recordings of noisy reverberant speech. BFE employs a very coarse model of the acoustic impulse response (AIR) from the source to the microphone, which has been shown to be effective if the speech to be recognized has been generated by artificially convolving nonreverberant speech with a constant AIR. Here we demonstrate that the model is also appropriate to be used in feature enhancement of true recordings of noisy reverberant speech. On the Multi-Channel Wall Street Journal Audio Visual corpus (MC-WSJ-AV) the word error rate is cut in half to 41.9 percent compared to the ETSI Standard Front-End using as input the signal of a single distant microphone with a single recognition pass.}},
  author       = {{Krueger, Alexander and Walter, Oliver and Leutnant, Volker and Haeb-Umbach, Reinhold}},
  booktitle    = {{Proc. Interspeech}},
  title        = {{{Bayesian Feature Enhancement for ASR of Noisy Reverberant Real-World Data}}},
  year         = {{2012}},
}

@article{11863,
  abstract     = {{In this contribution, a new observation model for the joint compensation of reverberation and noise in the logarithmic mel power spectral density domain will be considered. The proposed observation model relates the noisy reverberant feature to the underlying sequence of clean speech features and the feature of the noise. Nevertheless, due to the complex interaction of these variables in the target domain, the observationmodel cannot be applied to Bayesian feature enhancement directly, calling for approximations that eventually render the observation model useful. The performance of the approximated observation model will highly depend on the capability of modeling the difference between the model and the noisy reverberant observation. A detailed analysis of this observation error will be provided in this work. Among others, it will point out the need to account for the instantaneous ratio of the reverberant speech power and the noise power. Index Terms: Bayesian feature enhancement, observation model for noisy reverberant speech}},
  author       = {{Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  journal      = {{Speech Communication; 10. ITG Symposium; Proceedings of}},
  pages        = {{1--4}},
  title        = {{{Investigations Into a Statistical Observation Model for Logarithmic Mel Power Spectral Density Features of Noisy Reverberant Speech}}},
  year         = {{2012}},
}

@inproceedings{11864,
  abstract     = {{In this work, an observation model for the joint compensation of noise and reverberation in the logarithmic mel power spectral density domain is considered. It relates the features of the noisy reverberant speech to those of the non-reverberant speech and the noise. In contrast to enhancement of features only corrupted by reverberation (reverberant features), enhancement of noisy reverberant features requires a more sophisticated model for the error introduced by the proposed observation model. In a first consideration, it will be shown that this error is highly dependent on the instantaneous ratio of the power of reverberant speech to the power of the noise and, moreover, sensitive to the phase between reverberant speech and noise in the short-time discrete Fourier domain. Afterwards, a statistically motivated approach will be presented allowing for the model of the observation error to be inferred from the error model previously used for the reverberation only case. Finally, the developed observation error model will be utilized in a Bayesian feature enhancement scheme, leading to improvements in word accuracy on the AURORA5 database.}},
  author       = {{Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{Signal Processing, Communications and Computing (ICSPCC), 2012 IEEE International Conference on}},
  keywords     = {{Robust Automatic Speech Recognition, Bayesian feature enhancement, observation model for reverberant and noisy speech}},
  title        = {{{A Statistical Observation Model For Noisy Reverberant Speech Features and its Application to Robust ASR}}},
  year         = {{2012}},
}

@techreport{11865,
  author       = {{Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  title        = {{{Derivation of the Power Compensation Constant in the Observation Model for Reverberant Speech in the Logarithmic Mel Power Spectral Domain}}},
  year         = {{2012}},
}

@inproceedings{11910,
  author       = {{Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{International Workshop on Acoustic Signal Enhancement (IWAENC2012)}},
  title        = {{{Exploiting Temporal Correlations in Joint Multichannel Speech Separation and Noise Suppression using Hidden Markov Models}}},
  year         = {{2012}},
}

@inproceedings{11833,
  abstract     = {{In this paper we propose an approach to retrieve the geometry of an acoustic sensor network consisting of spatially distributed microphone arrays from unconstrained speech input. The calibration relies on Direction of Arrival (DoA) measurements which do not require a clock synchronization among the sensor nodes. The calibration problem is formulated as a cost function optimization task, which minimizes the squared differences between measured and predicted observations and additionally avoids the existence of minima that correspond to mirrored versions of the actual sensor orientations. Further, outlier measurements caused by reverberation are mitigated by a Random Sample Consensus (RANSAC) approach. The experimental results show a mean positioning error of at most 25 cm even in highly reverberant environments.}},
  author       = {{Jacob, Florian and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{International Workshop on Acoustic Signal Enhancement (IWAENC 2012)}},
  keywords     = {{Unsupervised, geometry calibration, microphone arrays, position self-calibration}},
  title        = {{{Microphone Array Position Self-Calibration from Reverberant Speech Input}}},
  year         = {{2012}},
}

@inproceedings{11925,
  abstract     = {{In this paper we present a system for car navigation by fusing sensor data on an Android smartphone. The key idea is to use both the internal sensors of the smartphone (e.g., gyroscope) and sensor data from the car (e.g., speed information) to support navigation via GPS. To this end we employ a CAN-Bus-to-Bluetooth adapter to establish a wireless connection between the smartphone and the CAN-Bus of the car. On the smartphone a strapdown algorithm and an error-state Kalman filter are used to fuse the different sensor data streams. The experimental results show that the system is able to maintain higher positioning accuracy during GPS dropouts, thus improving the availability and reliability, compared to GPS-only solutions.}},
  author       = {{Walter, Oliver and Schmalenstroeer, Joerg and Engler, Andreas and Haeb-Umbach, Reinhold}},
  booktitle    = {{9th Workshop on Positioning Navigation and Communication (WPNC 2012)}},
  keywords     = {{Smartphone, navigation, sensor fusion}},
  title        = {{{Smartphone-Based Sensor Fusion for Improved Vehicular Navigation}}},
  year         = {{2012}},
}

@inproceedings{11721,
  author       = {{Bevermeier, Maik and Flanke, Stephan and Haeb-Umbach, Reinhold and Stehr, Jan}},
  booktitle    = {{International Workshop on Intelligent Transportation (WIT 2011)}},
  title        = {{{A Platform for efficient Supply Chain Management Support in Logistics}}},
  year         = {{2011}},
}

@inbook{11774,
  abstract     = {{In this contribution classification rules for HMM-based speech recognition in the presence of a mismatch between training and test data are presented. The observed feature vectors are regarded as corrupted versions of underlying and unobservable clean feature vectors, which have the same statistics as the training data. Optimal classification then consists of two steps. First, the posterior density of the clean feature vector, given the observed feature vectors, has to be determined, and second, this posterior is employed in a modified classification rule, which accounts for imperfect estimates. We discuss different variants of the classification rule and further elaborate on the estimation of the clean speech feature posterior, using conditional Bayesian estimation. It is shown that this concept is fairly general and can be applied to different scenarios, such as noisy or reverberant speech recognition.}},
  author       = {{Haeb-Umbach, Reinhold}},
  booktitle    = {{Robust Speech Recognition of Uncertain or Missing Data}},
  editor       = {{Haeb-Umbach, Reinhold and Kolossa, Dorothea}},
  publisher    = {{Springer}},
  title        = {{{Uncertainty Decoding and Conditional Bayesian Estimation}}},
  year         = {{2011}},
}

@inbook{11775,
  author       = {{Haeb-Umbach, Reinhold}},
  booktitle    = {{Baustelle Informationsgesellschaft und Universität heute}},
  publisher    = {{Ferdinand Schoeningh Verlag, Paderborn}},
  title        = {{{Können Computer sprechen und hören, sollen sie es überhaupt können? Sprachverarbeitung und ambiente Intelligenz}}},
  year         = {{2011}},
}

@article{11807,
  author       = {{Herbig, Tobias and Gerl, Franz and Minker, Wolfgang and Haeb-Umbach, Reinhold}},
  journal      = {{Evolving Systems}},
  number       = {{3}},
  pages        = {{199--214}},
  title        = {{{Adaptive Systems for Unsupervised Speaker Tracking and Speech Recognition}}},
  volume       = {{2}},
  year         = {{2011}},
}

@inbook{11843,
  abstract     = {{Employing automatic speech recognition systems in hands-free communication applications is accompanied by perfomance degradation due to background noise and, in particular, due to reverberation. These two kinds of distortion alter the shape of the feature vector trajectory extracted from the microphone signal and consequently lead to a discrepancy between training and testing conditions for the recognizer. In this chapter we present a feature enhancement approach aiming at the joint compensation of noise and reverberation to improve the performance by restoring the training conditions. For the enhancement we concentrate on the logarithmic mel power spectral coefficients as features, which are computed at an intermediate stage to obtain the widely used mel frequency cepstral coefficients. The proposed technique is based on a Bayesian framework, to attempt to infer the posterior distribution of the clean features given the observation of all past corrupted features. It exploits information from a priori models describing the dynamics of clean speech and noise-only feature vector trajectories as well as from an observation model relating the reverberant noisy to the clean features. The observation model relies on a simplified stochastic model of the room impulse response (RIR) between the speaker and the microphone, having only two parameters, namely RIR energy and reverberation time, which can be estimated from the captured microphone signal. The performance of the proposed enhancement technique is finally experimentally studied by means of recognition accuracy obtained for a connected digits recognition task under different noise and reverberation conditions using the Aurora~5 database.}},
  author       = {{Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{Robust Speech Recognition of Uncertain or Missing Data}},
  editor       = {{Haeb-Umbach, Reinhold and Kolossa, Dorothea}},
  publisher    = {{Springer}},
  title        = {{{A Model-Based Approach to Joint Compensation of Noise and Reverberation for Speech Recognition}}},
  year         = {{2011}},
}

@inproceedings{11845,
  abstract     = {{The paper proposes a modification of the standard maximum a posteriori (MAP) method for the estimation of the parameters of a Gaussian process for cases where the process is superposed by additive Gaussian observation errors of known variance. Simulations on artificially generated data demonstrate the superiority of the proposed method. While reducing to the ordinary MAP approach in the absence of observation noise, the improvement becomes the more pronounced the larger the variance of the observation noise. The method is further extended to track the parameters in case of non-stationary Gaussian processes.}},
  author       = {{Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2011)}},
  keywords     = {{Gaussian processes, MAP-based estimation, maximum a posteriori method, maximum likelihood estimation, nonstationary Gaussian processes}},
  pages        = {{3596--3599}},
  title        = {{{MAP-based estimation of the parameters of non-stationary Gaussian processes from noisy observations}}},
  doi          = {{10.1109/ICASSP.2011.5946256}},
  year         = {{2011}},
}

@article{11850,
  abstract     = {{In this paper, we present a novel blocking matrix and fixed beamformer design for a generalized sidelobe canceler for speech enhancement in a reverberant enclosure. They are based on a new method for estimating the acoustical transfer function ratios in the presence of stationary noise. The estimation method relies on solving a generalized eigenvalue problem in each frequency bin. An adaptive eigenvector tracking utilizing the power iteration method is employed and shown to achieve a high convergence speed. Simulation results demonstrate that the proposed beamformer leads to better noise and interference reduction and reduced speech distortions compared to other blocking matrix designs from the literature.}},
  author       = {{Krueger, Alexander and Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{acoustical transfer function ratio, adaptive eigenvector tracking, array signal processing, beamformer design, blocking matrix, eigenvalues and eigenfunctions, eigenvector-based transfer function ratios estimation, generalized sidelobe canceler, interference reduction, iterative methods, power iteration method, reduced speech distortions, reverberant enclosure, reverberation, speech enhancement, stationary noise}},
  number       = {{1}},
  pages        = {{206--219}},
  title        = {{{Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation}}},
  doi          = {{10.1109/TASL.2010.2047324}},
  volume       = {{19}},
  year         = {{2011}},
}

@inbook{11856,
  abstract     = {{In this contribution, conditional Bayesian estimation employing a phase-sensitive observation model for noise robust speech recognition will be studied. After a review of speech recognition under the presence of corrupted features, termed uncertainty decoding, the estimation of the posterior distribution of the uncorrupted (clean) feature vector will be shown to be a key element of noise robust speech recognition. The estimation process will be based on three major components: an a priori model of the unobservable data, an observation model relating the unobservable data to the corrupted observation and an inference algorithm, finally allowing for a computationally tractable solution. Special stress will be laid on a detailed derivation of the phase-sensitive observation model and the required moments of the phase factor distribution. Thereby, it will not only be proven analytically that the phase factor distribution is non-Gaussian but also that all central moments can (approximately) be computed solely based on the used mel filter bank, finally rendering the moments independent of noise type and signal-to-noise ratio. The phase-sensitive observation model will then be incorporated into a model-based feature enhancement scheme and recognition experiments will be carried out on the Aurora~2 and Aurora~4 databases. The importance of incorporating phase factor information into the enhancement scheme is pointed out by all recognition results. Application of the proposed scheme under the derived uncertainty decoding framework further leads to significant improvements in both recognition tasks, eventually reaching the performance achieved with the ETSI advanced front-end.}},
  author       = {{Leutnant, Volker and Haeb-Umbach, Reinhold}},
  booktitle    = {{Robust Speech Recognition of Uncertain or Missing Data}},
  editor       = {{Haeb-Umbach, Reinhold and Kolossa, Dorothea}},
  publisher    = {{Springer}},
  title        = {{{Conditional Bayesian Estimation Employing a Phase-Sensitive Observation Model for Noise Robust Speech Recognition}}},
  year         = {{2011}},
}

@inproceedings{11866,
  abstract     = {{In this work, a splitting and weighting scheme that allows for splitting a Gaussian density into a Gaussian mixture density (GMM) is extended to allow the mixture components to be arranged along arbitrary directions. The parameters of the Gaussian mixture are chosen such that the GMM and the original Gaussian still exhibit equal central moments up to an order of four. The resulting mixtures{\rq} covariances will have eigenvalues that are smaller than those of the covariance of the original distribution, which is a desirable property in the context of non-linear state estimation, since the underlying assumptions of the extended K ALMAN filter are better justified in this case. Application to speech feature enhancement in the context of noise-robust automatic speech recognition reveals the beneficial properties of the proposed approach in terms of a reduced word error rate on the Aurora 2 recognition task.}},
  author       = {{Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2011}},
  title        = {{{A versatile Gaussian splitting approach to non-linear state estimation and its application to noise-robust ASR}}},
  year         = {{2011}},
}

@inproceedings{11911,
  abstract     = {{In this paper we address the problem of initial seed selection for frequency domain iterative blind speech separation (BSS) algorithms. The derivation of the seeding algorithm is guided by the goal to select samples which are likely to be caused by source activity and not by noise and at the same time originate from different sources. The proposed algorithm has moderate computational complexity and finds better seed values than alternative schemes, as is demonstrated by experiments on the database of the SiSEC2010 challenge.}},
  author       = {{Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2011}},
  title        = {{{On Initial Seed Selection for Frequency Domain Blind Speech Separation}}},
  year         = {{2011}},
}

@book{11945,
  editor       = {{Kolossa, Dorothea and Haeb-Umbach, Reinhold}},
  publisher    = {{Springer}},
  title        = {{{Robust Speech Recognition of Uncertain or Missing Data --- Theory and Applications}}},
  year         = {{2011}},
}

@inproceedings{11889,
  abstract     = {{In this paper we propose to jointly consider Segmental Dynamic Time Warping and distance clustering for the unsupervised learning of acoustic events. As a result, the computational complexity increases only linearly with the dababase size compared to a quadratic increase in a sequential setup, where all pairwise SDTW distances between segments are computed prior to clustering. Further, we discuss options for seed value selection for clustering and show that drawing seeds with a probability proportional to the distance from the already drawn seeds, known as K-means++ clustering, results in a significantly higher probability of finding representatives of each of the underlying classes, compared to the commonly used draws from a uniform distribution. Experiments are performed on an acoustic event classification and an isolated digit recognition task, where on the latter the final word accuracy approaches that of supervised training.}},
  author       = {{Schmalenstroeer, Joerg and Bartek, Markus and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2011}},
  title        = {{{Unsupervised learning of acoustic events using dynamic time warping and hierarchical K-means++ clustering}}},
  year         = {{2011}},
}

@inproceedings{11896,
  abstract     = {{In this paper we propose a procedure for estimating the geometric configuration of an arbitrary acoustic sensor placement. It determines the position and the orientation of microphone arrays in 2D while locating a source by direction-of-arrival (DoA) estimation. Neither artificial calibration signals nor unnatural user activity are required. The problem of scale indeterminacy inherent to DoA-only observations is solved by adding time difference of arrival (TDOA) measurements. The geometry calibration method is numerically stable and delivers precise results in moderately reverberated rooms. Simulation results are confirmed by laboratory experiments.}},
  author       = {{Schmalenstroeer, Joerg and Jacob, Florian and Haeb-Umbach, Reinhold and Hennecke, Marius and Fink, Gernot A.}},
  booktitle    = {{Interspeech 2011}},
  title        = {{{Unsupervised Geometry Calibration of Acoustic Sensor Networks Using Source Correspondences}}},
  year         = {{2011}},
}

