@inproceedings{11716,
  abstract     = {{The accuracy of automatic speech recognition systems in noisy and reverberant environments can be improved notably by exploiting the uncertainty of the estimated speech features using so-called uncertainty-of-observation techniques. In this paper, we introduce a new Bayesian decision rule that can serve as a mathematical framework from which both known and new uncertainty-of-observation techniques can be either derived or approximated. The new decision rule in its direct form leads to the new significance decoding approach for Gaussian mixture models, which results in better performance compared to standard uncertainty-of-observation techniques in different additive and convolutive noise scenarios.}},
  author       = {{Abdelaziz, Ahmed H. and Zeiler, Steffen and Kolossa, Dorothea and Leutnant, Volker and Haeb-Umbach, Reinhold}},
  booktitle    = {{Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on}},
  issn         = {{1520-6149}},
  keywords     = {{Bayes methods, Gaussian processes, convolution, decision theory, decoding, noise, reverberation, speech coding, speech recognition, Bayesian decision rule, GMM, Gaussian mixture models, additive noise scenarios, automatic speech recognition systems, convolutive noise scenarios, decoding approach, mathematical framework, reverberant environments, significance decoding, speech feature estimation, uncertainty-of-observation techniques, Hidden Markov models, Maximum likelihood decoding, Noise, Speech, Speech recognition, Uncertainty, Uncertainty-of-observation, modified imputation, noise robust speech recognition, significance decoding, uncertainty decoding}},
  pages        = {{6827--6831}},
  title        = {{{GMM-based significance decoding}}},
  doi          = {{10.1109/ICASSP.2013.6638984}},
  year         = {{2013}},
}

@inproceedings{11740,
  abstract     = {{In this contribution we derive the Maximum A-Posteriori (MAP) estimates of the parameters of a Gaussian Mixture Model (GMM) in the presence of noisy observations. We assume the distortion to be white Gaussian noise of known mean and variance. An approximate conjugate prior of the GMM parameters is derived allowing for a computationally efficient implementation in a sequential estimation framework. Simulations on artificially generated data demonstrate the superiority of the proposed method compared to the Maximum Likelihood technique and to the ordinary MAP approach, whose estimates are corrected by the known statistics of the distortion in a straightforward manner.}},
  author       = {{Chinaev, Aleksej and Haeb-Umbach, Reinhold}},
  booktitle    = {{38th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2013)}},
  issn         = {{1520-6149}},
  keywords     = {{Gaussian noise, maximum likelihood estimation, parameter estimation, GMM parameter, Gaussian mixture model, MAP estimation, Map-based estimation, maximum a-posteriori estimation, maximum likelihood technique, noisy observation, sequential estimation framework, white Gaussian noise, Additive noise, Gaussian mixture model, Maximum likelihood estimation, Noise measurement, Gaussian mixture model, Maximum a posteriori estimation, Maximum likelihood estimation}},
  pages        = {{3352--3356}},
  title        = {{{MAP-based Estimation of the Parameters of a Gaussian Mixture Model in the Presence of Noisy Observations}}},
  doi          = {{10.1109/ICASSP.2013.6638279}},
  year         = {{2013}},
}

@inproceedings{11742,
  abstract     = {{In this paper we present an improved version of the recently proposed Maximum A-Posteriori (MAP) based noise power spectral density estimator. An empirical bias compensation and bandwidth adjustment reduce bias and variance of the noise variance estimates. The main advantage of the MAP-based postprocessor is its low estimation variance. The estimator is employed in the second stage of a two-stage single-channel speech enhancement system, where eight different state-of-the-art noise tracking algorithms were tested in the first stage. While the postprocessor hardly affects the results in stationary noise scenarios, it becomes the more effective the more nonstationary the noise is. The proposed postprocessor was able to improve all systems in babble noise w.r.t. the perceptual evaluation of speech quality performance.}},
  author       = {{Chinaev, Aleksej and Haeb-Umbach, Reinhold and Taghia, Jalal and Martin, Rainer}},
  booktitle    = {{38th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2013)}},
  issn         = {{1520-6149}},
  pages        = {{7477--7481}},
  title        = {{{Improved Single-Channel Nonstationary Noise Tracking by an Optimized MAP-based Postprocessor}}},
  doi          = {{10.1109/ICASSP.2013.6639116}},
  year         = {{2013}},
}

@inproceedings{11762,
  abstract     = {{Among the different configurations of multi-microphone systems, e.g., in applications of speech dereverberation or denoising, we consider the case without a priori information of the microphone-array geometry. This naturally invokes explicit or implicit identification of source-receiver transfer functions as an indirect description of the microphone-array configuration. However, this blind channel identification (BCI) has been difficult due to the lack of unique identifiability in the presence of observation noise or near-common channel zeros. In this paper, we study the implicit BCI performance of blind signal enhancement techniques such as the adaptive principal component analysis (PCA) or the iterative blind equalization and channel identification (BENCH). To this end, we make use of a recently proposed metric, the normalized filter-projection misalignment (NFPM), which is tailored for BCI evaluation in ill-conditioned (e.g., noisy) scenarios. The resulting understanding of implicit BCI performance can help to judge the behavior of multi-microphone speech enhancement systems and the suitability of implicit BCI to serve channel-based (i.e., channel-informed) enhancement.}},
  author       = {{Enzner, Gerald and Schmid, Dominic and Haeb-Umbach, Reinhold}},
  booktitle    = {{21th European Signal Processing Conference (EUSIPCO 2013)}},
  title        = {{{On the Acoustic Channel Identification in Multi-Microphone Systems via Adaptive Blind Signal Enhancement Techniques}}},
  year         = {{2013}},
}

@inproceedings{11815,
  author       = {{Heymann, Jahn and Walter, Oliver and Haeb-Umbach, Reinhold and Raj, Bhiksha}},
  booktitle    = {{Automatic Speech Recognition and Understanding Workshop (ASRU 2013)}},
  title        = {{{Unsupervised Word Segmentation from Noisy Input}}},
  year         = {{2013}},
}

@inproceedings{11816,
  abstract     = {{In this paper, we consider the Maximum Likelihood (ML) estimation of the parameters of a GAUSSIAN in the presence of censored, i.e., clipped data. We show that the resulting Expectation Maximization (EM) algorithm delivers virtually biasfree and efficient estimates, and we discuss its convergence properties. We also discuss optimal classification in the presence of censored data. Censored data are frequently encountered in wireless LAN positioning systems based on the fingerprinting method employing signal strength measurements, due to the limited sensitivity of the portable devices. Experiments both on simulated and real-world data demonstrate the effectiveness of the proposed algorithms.}},
  author       = {{Hoang, Manh Kha and Haeb-Umbach, Reinhold}},
  booktitle    = {{38th International Conference on Acoustics, Speech, and Signal Processing (ICASSP 2013)}},
  issn         = {{1520-6149}},
  keywords     = {{Gaussian processes, Global Positioning System, convergence, expectation-maximisation algorithm, fingerprint identification, indoor radio, signal classification, wireless LAN, EM algorithm, ML estimation, WiFi indoor positioning, censored Gaussian data classification, clipped data, convergence properties, expectation maximization algorithm, fingerprinting method, maximum likelihood estimation, optimal classification, parameters estimation, portable devices sensitivity, signal strength measurements, wireless LAN positioning systems, Convergence, IEEE 802.11 Standards, Maximum likelihood estimation, Parameter estimation, Position measurement, Training, Indoor positioning, censored data, expectation maximization, signal strength, wireless LAN}},
  pages        = {{3721--3725}},
  title        = {{{Parameter estimation and classification of censored Gaussian data with application to WiFi indoor positioning}}},
  doi          = {{10.1109/ICASSP.2013.6638353}},
  year         = {{2013}},
}

@inproceedings{11841,
  abstract     = {{Recently, substantial progress has been made in the field of reverberant speech signal processing, including both single- and multichannel de-reverberation techniques, and automatic speech recognition (ASR) techniques robust to reverberation. To evaluate state-of-the-art algorithms and obtain new insights regarding potential future research directions, we propose a common evaluation framework including datasets, tasks, and evaluation metrics for both speech enhancement and ASR techniques. The proposed framework will be used as a common basis for the REVERB (REverberant Voice Enhancement and Recognition Benchmark) challenge. This paper describes the rationale behind the challenge, and provides a detailed description of the evaluation framework and benchmark results.}},
  author       = {{Kinoshita, Keisuke and Delcroix, Marc and Yoshioka, Takuya and Nakatani, Tomohiro and Habets, Emanuel and Haeb-Umbach, Reinhold and Leutnant, Volker and Sehr, Armin and Kellermann, Walter and Maas, Roland and Gannot, Sharon and Raj, Bhiksha}},
  booktitle    = {{ IEEE Workshop on Applications of Signal Processing to Audio and Acoustics }},
  keywords     = {{Reverberant speech, dereverberation, ASR, evaluation, challenge}},
  pages        = {{ 22--23 }},
  title        = {{{The reverb challenge: a common evaluation framework for dereverberation and recognition of reverberant speech}}},
  year         = {{2013}},
}

@article{11862,
  abstract     = {{In this contribution we extend a previously proposed Bayesian approach for the enhancement of reverberant logarithmic mel power spectral coefficients for robust automatic speech recognition to the additional compensation of background noise. A recently proposed observation model is employed whose time-variant observation error statistics are obtained as a side product of the inference of the a posteriori probability density function of the clean speech feature vectors. Further a reduction of the computational effort and the memory requirements are achieved by using a recursive formulation of the observation model. The performance of the proposed algorithms is first experimentally studied on a connected digits recognition task with artificially created noisy reverberant data. It is shown that the use of the time-variant observation error model leads to a significant error rate reduction at low signal-to-noise ratios compared to a time-invariant model. Further experiments were conducted on a 5000 word task recorded in a reverberant and noisy environment. A significant word error rate reduction was obtained demonstrating the effectiveness of the approach on real-world data.}},
  author       = {{Leutnant, Volker and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{Bayes methods, compensation, error statistics, reverberation, speech recognition, Bayesian feature enhancement, background noise, clean speech feature vectors, compensation, connected digits recognition task, error statistics, memory requirements, noisy reverberant data, posteriori probability density function, recursive formulation, reverberant logarithmic mel power spectral coefficients, robust automatic speech recognition, signal-to-noise ratios, time-variant observation, word error rate reduction, Robust automatic speech recognition, model-based Bayesian feature enhancement, observation model for reverberant and noisy speech, recursive observation model}},
  number       = {{8}},
  pages        = {{1640--1652}},
  title        = {{{Bayesian Feature Enhancement for Reverberation and Noise Robust Speech Recognition}}},
  doi          = {{10.1109/TASL.2013.2258013}},
  volume       = {{21}},
  year         = {{2013}},
}

@inproceedings{11909,
  abstract     = {{We present a novel method to exploit correlations of adjacent time-frequency (TF)-slots for a sparseness-based blind speech separation (BSS) system. Usually, these correlations are exploited by some heuristic smoothing techniques in the post-processing of the estimated soft TF masks. We propose a different approach: Based on our previous work with one-dimensional (1D)-hidden Markov models (HMMs) along the time axis we extend the modeling to two-dimensional (2D)-HMMs to exploit both temporal and spectral correlations in the speech signal. Based on the principles of turbo decoding we solved the complex inference of 2D-HMMs by a modified forward-backward algorithm which operates alternatingly along the time and the frequency axis. Extrinsic information is exchanged between these steps such that increasingly better soft time-frequency masks are obtained, leading to improved speech separation performance in highly reverberant recording conditions.}},
  author       = {{Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{21th European Signal Processing Conference (EUSIPCO 2013)}},
  title        = {{{Blind Speech Separation Exploiting Temporal and Spectral Correlations Using Turbo Decoding of 2D-HMMs}}},
  year         = {{2013}},
}

@inproceedings{11917,
  abstract     = {{In this paper we present a speech presence probability (SPP) estimation algorithmwhich exploits both temporal and spectral correlations of speech. To this end, the SPP estimation is formulated as the posterior probability estimation of the states of a two-dimensional (2D) Hidden Markov Model (HMM). We derive an iterative algorithm to decode the 2D-HMM which is based on the turbo principle. The experimental results show that indeed the SPP estimates improve from iteration to iteration, and further clearly outperform another state-of-the-art SPP estimation algorithm.}},
  author       = {{Vu, Dang Hai Tran and Haeb-Umbach, Reinhold}},
  booktitle    = {{38th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2013)}},
  issn         = {{1520-6149}},
  keywords     = {{correlation methods, estimation theory, hidden Markov models, iterative methods, probability, spectral analysis, speech processing, 2D HMM, SPP estimates, iterative algorithm, posterior probability estimation, spectral correlation, speech presence probability estimation, state-of-the-art SPP estimation algorithm, temporal correlation, turbo principle, two-dimensional hidden Markov model, Correlation, Decoding, Estimation, Iterative decoding, Noise, Speech, Vectors}},
  pages        = {{863--867}},
  title        = {{{Using the turbo principle for exploiting temporal and spectral correlations in speech presence probability estimation}}},
  doi          = {{10.1109/ICASSP.2013.6637771}},
  year         = {{2013}},
}

@inproceedings{11921,
  abstract     = {{In this paper we consider the unsupervised word discovery from phonetic input. We employ a word segmentation algorithm which simultaneously develops a lexicon, i.e., the transcription of a word in terms of a phone sequence, learns a n-gram language model describing word and word sequence probabilities, and carries out the segmentation itself. The underlying statistical model is that of a Pitman-Yor process, a concept known from Bayesian non-parametrics, which allows for an a priori unknown and unlimited number of different words. Using a hierarchy of Pitman-Yor processes, language models of different order can be employed and nesting it with another hierarchy of Pitman-Yor processes on the phone level allows for backing off unknown word unigrams by phone m-grams. We present results on a large-vocabulary task, assuming an error-free phone sequence is given. We finish by discussing options how to cope with noisy phone sequences.}},
  author       = {{Walter, Oliver and Haeb-Umbach, Reinhold and Chaudhuri, Sourish and Raj, Bhiksha}},
  booktitle    = {{IEEE International Conference on Robotics and Automation (ICRA 2013)}},
  title        = {{{Unsupervised Word Discovery from Phonetic Input Using Nested Pitman-Yor Language Modeling}}},
  year         = {{2013}},
}

@inproceedings{11924,
  author       = {{Walter, Oliver and Korthals, Timo and Haeb-Umbach, Reinhold and Raj, Bhiksha}},
  booktitle    = {{Automatic Speech Recognition and Understanding Workshop (ASRU 2013)}},
  title        = {{{Hierarchical System for Word Discovery Exploiting DTW-Based Initialization}}},
  year         = {{2013}},
}

@techreport{11926,
  abstract     = {{In this paper we present a novel initialization method for unsupervised learning of acoustic patterns in recordings of continuous speech. The pattern discovery task is solved by dynamic time warping whose performance we improve by a smart starting point selection. This enables a more accurate discovery of patterns compared to conventional approaches. After graph-based clustering the patterns are employed for training hidden Markov models for an unsupervised speech acquisition. By iterating between model training and decoding in an EM-like framework the word accuracy is continuously improved. On the TIDIGITS corpus we achieve a word error rate of about 13 percent by the proposed unsupervised pattern discovery approach, which neither assumes knowledge of the acoustic units nor of the labels of the training data.}},
  author       = {{Walter, Oliver and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  title        = {{{A Novel Initialization Method for Unsupervised Learning of Acoustic Patterns in Speech (FGNT-2013-01)}}},
  year         = {{2013}},
}

@inproceedings{11832,
  abstract     = {{In this paper we propose an approach to retrieve the absolute geometry of an acoustic sensor network, consisting of spatially distributed microphone arrays, from reverberant speech input. The calibration relies on direction of arrival measurements of the individual arrays. The proposed calibration algorithm is derived from a maximum-likelihood approach employing circular statistics. Since a sensor node consists of a microphone array with known intra-array geometry, we are able to obtain an absolute geometry estimate, including angles and distances. Simulation results demonstrate the effectiveness of the approach.}},
  author       = {{Jacob, Florian and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{38th International Conference on Acoustics, Speech, and Signal Processing (ICASSP 2013)}},
  issn         = {{1520-6149}},
  keywords     = {{Geometry calibration, microphone arrays, position self-calibration}},
  pages        = {{116--120}},
  title        = {{{DoA-Based Microphone Array Position Self-Calibration Using Circular Statistic}}},
  doi          = {{10.1109/ICASSP.2013.6637620}},
  year         = {{2013}},
}

@inproceedings{11891,
  abstract     = {{In this paper we present a combined hardware/software approach for synchronizing the sampling clocks of an acoustic sensor network. A first clock frequency offset estimate is obtained by a time stamp exchange protocol with a low data rate and computational requirements. The estimate is then postprocessed by a Kalman filter which exploits the specific properties of the statistics of the frequency offset estimation error. In long term experiments the deviation between the sampling oscillators of two sensor nodes never exceeded half a sample with a wired and with a wireless link between the nodes. The achieved precision enables the estimation of time difference of arrival values across different hardware devices without sharing a common sampling hardware.}},
  author       = {{Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{21th European Signal Processing Conference (EUSIPCO 2013)}},
  keywords     = {{synchronization, acoustic sensor network}},
  title        = {{{Sampling Rate Synchronisation in Acoustic Sensor Networks with a Pre-Trained Clock Skew Error Model}}},
  year         = {{2013}},
}

@inproceedings{11818,
  abstract     = {{In this paper we present a system for indoor navigation based on received signal strength index information of Wireless-LAN access points and relative position estimates. The relative position information is gathered from inertial smartphone sensors using a step detection and an orientation estimate. Our map data is hosted on a server employing a map renderer and a SQL database. The database includes a complete multilevel office building, within which the user can navigate. During navigation, the client retrieves the position estimate from the server, together with the corresponding map tiles to visualize the user's position on the smartphone display.}},
  author       = {{Hoang, Manh Kha and Schmitz, Sarah and Drueke, Christian and Vu, Dang Hai Tran and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{Positioning Navigation and Communication (WPNC), 2013 10th Workshop on}},
  keywords     = {{SQL, navigation, smart phones, wireless LAN, RSSI, SQL database, complete multilevel office building, inertial sensor information, inertial smartphone sensors, map renderer, received signal strength index information, relative position estimates, server based indoor navigation, step detection, wireless-LAN access points, Smartphone, fingerprint, indoor navigation, map tile}},
  pages        = {{1--6}},
  title        = {{{Server based indoor navigation using RSSI and inertial sensor information}}},
  doi          = {{10.1109/WPNC.2013.6533263}},
  year         = {{2013}},
}

@inproceedings{11817,
  abstract     = {{In this paper we present a modified hidden Markov model (HMM) for the fusion of received signal strength index (RSSI) information of WiFi access points and relative position information which is obtained from the inertial sensors of a smartphone for indoor positioning. Since the states of the HMM represent the potential user locations, their number determines the quantization error introduced by discretizing the allowable user positions through the use of the HMM. To reduce this quantization error we introduce â??pseudoâ?? states, whose emission probability, which models the RSSI measurements at this location, is synthesized from those of the neighboring states of which a Gaussian emission probability has been estimated during the training phase. The experimental results demonstrate the effectiveness of this approach. By introducing on average two pseudo states per original HMM state the positioning error could be significantly reduced without increasing the training effort.}},
  author       = {{Hoang, Manh Kha and Schmalenstroeer, Joerg and Drueke, Christian and Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{21th European Signal Processing Conference (EUSIPCO 2013)}},
  title        = {{{A Hidden Markov Model for Indoor User Tracking Based on WiFi Fingerprinting and Step Detection}}},
  year         = {{2013}},
}

@inproceedings{11741,
  author       = {{Chinaev, Aleksej and Haeb-Umbach, Reinhold}},
  booktitle    = {{Speech Communication; 10. ITG Symposium; Proceedings.}},
  title        = {{{Quality Analysis and Optimization of the MAP-based Noise Power Spectral Density Tracker}}},
  year         = {{2012}},
}

@inproceedings{11745,
  abstract     = {{In this paper we present a novel noise power spectral density tracking algorithm and its use in single-channel speech enhancement. It has the unique feature that it is able to track the noise statistics even if speech is dominant in a given time-frequency bin. As a consequence it can follow non-stationary noise superposed by speech, even in the critical case of rising noise power. The algorithm requires an initial estimate of the power spectrum of speech and is thus meant to be used as a postprocessor to a first speech enhancement stage. An experimental comparison with a state-of-the-art noise tracking algorithm demonstrates lower estimation errors under low SNR conditions and smaller fluctuations of the estimated values, resulting in improved speech quality as measured by PESQ scores.}},
  author       = {{Chinaev, Aleksej and Krueger, Alexander and Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{37th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2012)}},
  keywords     = {{MAP parameter estimation, noise power estimation, speech enhancement}},
  title        = {{{Improved Noise Power Spectral Density Tracking by a MAP-based Postprocessor}}},
  year         = {{2012}},
}

@inbook{11844,
  author       = {{Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{Techniques for Noise Robustness in Automatic Speech Recognition}},
  publisher    = {{Wiley}},
  title        = {{{Reverberant Speech Recognition}}},
  year         = {{2012}},
}

