@inproceedings{11739,
  abstract     = {{Noise tracking is an important component of speech enhancement algorithms. Of the many noise trackers proposed, Minimum Statistics (MS) is a particularly popular one due to its simple parameterization and at the same time excellent performance. In this paper we propose to further reduce the number of MS parameters by giving an alternative derivation of an optimal smoothing constant. At the same time the noise tracking performance is improved as is demonstrated by experiments employing speech degraded by various noise types and at different SNR values.}},
  author       = {{Chinaev, Aleksej and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2015}},
  keywords     = {{speech enhancement, noise tracking, optimal smoothing}},
  pages        = {{1785--1789}},
  title        = {{{On Optimal Smoothing in Minimum Statistics Based Noise Tracking}}},
  year         = {{2015}},
}

@inproceedings{11745,
  abstract     = {{In this paper we present a novel noise power spectral density tracking algorithm and its use in single-channel speech enhancement. It has the unique feature that it is able to track the noise statistics even if speech is dominant in a given time-frequency bin. As a consequence it can follow non-stationary noise superposed by speech, even in the critical case of rising noise power. The algorithm requires an initial estimate of the power spectrum of speech and is thus meant to be used as a postprocessor to a first speech enhancement stage. An experimental comparison with a state-of-the-art noise tracking algorithm demonstrates lower estimation errors under low SNR conditions and smaller fluctuations of the estimated values, resulting in improved speech quality as measured by PESQ scores.}},
  author       = {{Chinaev, Aleksej and Krueger, Alexander and Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{37th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2012)}},
  keywords     = {{MAP parameter estimation, noise power estimation, speech enhancement}},
  title        = {{{Improved Noise Power Spectral Density Tracking by a MAP-based Postprocessor}}},
  year         = {{2012}},
}

@article{11850,
  abstract     = {{In this paper, we present a novel blocking matrix and fixed beamformer design for a generalized sidelobe canceler for speech enhancement in a reverberant enclosure. They are based on a new method for estimating the acoustical transfer function ratios in the presence of stationary noise. The estimation method relies on solving a generalized eigenvalue problem in each frequency bin. An adaptive eigenvector tracking utilizing the power iteration method is employed and shown to achieve a high convergence speed. Simulation results demonstrate that the proposed beamformer leads to better noise and interference reduction and reduced speech distortions compared to other blocking matrix designs from the literature.}},
  author       = {{Krueger, Alexander and Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{acoustical transfer function ratio, adaptive eigenvector tracking, array signal processing, beamformer design, blocking matrix, eigenvalues and eigenfunctions, eigenvector-based transfer function ratios estimation, generalized sidelobe canceler, interference reduction, iterative methods, power iteration method, reduced speech distortions, reverberant enclosure, reverberation, speech enhancement, stationary noise}},
  number       = {{1}},
  pages        = {{206--219}},
  title        = {{{Speech Enhancement With a GSC-Like Structure Employing Eigenvector-Based Transfer Function Ratios Estimation}}},
  doi          = {{10.1109/TASL.2010.2047324}},
  volume       = {{19}},
  year         = {{2011}},
}

@inproceedings{11913,
  abstract     = {{In this paper we propose to employ directional statistics in a complex vector space to approach the problem of blind speech separation in the presence of spatially correlated noise. We interpret the values of the short time Fourier transform of the microphone signals to be draws from a mixture of complex Watson distributions, a probabilistic model which naturally accounts for spatial aliasing. The parameters of the density are related to the a priori source probabilities, the power of the sources and the transfer function ratios from sources to sensors. Estimation formulas are derived for these parameters by employing the Expectation Maximization (EM) algorithm. The E-step corresponds to the estimation of the source presence probabilities for each time-frequency bin, while the M-step leads to a maximum signal-to-noise ratio (MaxSNR) beamformer in the presence of uncertainty about the source activity. Experimental results are reported for an implementation in a generalized sidelobe canceller (GSC) like spatial beamforming configuration for 3 speech sources with significant coherent noise in reverberant environments, demonstrating the usefulness of the novel modeling framework.}},
  author       = {{Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2010)}},
  keywords     = {{array signal processing, blind source separation, blind speech separation, complex vector space, complex Watson distribution, directional statistics, expectation-maximisation algorithm, expectation maximization algorithm, Fourier transform, Fourier transforms, generalized sidelobe canceller, interference suppression, maximum signal-to-noise ratio beamformer, microphone signal, probabilistic model, spatial aliasing, spatial beamforming configuration, speech enhancement, statistical distributions}},
  pages        = {{241--244}},
  title        = {{{Blind speech separation employing directional statistics in an Expectation Maximization framework}}},
  doi          = {{10.1109/ICASSP.2010.5495994}},
  year         = {{2010}},
}

@article{11937,
  abstract     = {{In automatic speech recognition, hidden Markov models (HMMs) are commonly used for speech decoding, while switching linear dynamic models (SLDMs) can be employed for a preceding model-based speech feature enhancement. In this paper, these model types are combined in order to obtain a novel iterative speech feature enhancement and recognition architecture. It is shown that speech feature enhancement with SLDMs can be improved by feeding back information from the HMM to the enhancement stage. Two different feedback structures are derived. In the first, the posteriors of the HMM states are used to control the model probabilities of the SLDMs, while in the second they are employed to directly influence the estimate of the speech feature distribution. Both approaches lead to improvements in recognition accuracy both on the AURORA2 and AURORA4 databases compared to non-iterative speech feature enhancement with SLDMs. It is also shown that a combination with uncertainty decoding further enhances performance.}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{AURORA2 databases, AURORA4 databases, automatic speech recognition, feedback structures, hidden Markov models, HMM, iterative methods, iterative speech feature enhancement, model probabilities, speech decoding, speech enhancement, speech feature distribution, speech recognition, switching linear dynamic models}},
  number       = {{5}},
  pages        = {{974--984}},
  title        = {{{Approaches to Iterative Speech Feature Enhancement and Recognition}}},
  doi          = {{10.1109/TASL.2009.2014894}},
  volume       = {{17}},
  year         = {{2009}},
}

@inproceedings{11935,
  abstract     = {{The generalized sidelobe canceller by Griffith and Jim is a robust beamforming method to enhance a desired (speech) signal in the presence of stationary noise. Its performance depends to a high degree on the construction of the blocking matrix which produces noise reference signals for the subsequent adaptive interference canceller. Especially in reverberated environments the beamformer may suffer from signal leakage and reduced noise suppression. In this paper a new blocking matrix is proposed. It is based on a generalized eigenvalue problem whose solution provides an indirect estimation of the transfer functions from the source to the sensors. The quality of the new generalized eigenvector blocking matrix is studied in simulated rooms with different reverberation times and is compared to alternatives proposed in the literature.}},
  author       = {{Warsitz, Ernst and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}},
  keywords     = {{adaptive interference canceller, adaptive signal processing, array signal processing, beamforming method, eigenvalues and eigenfunctions, generalized eigenvector blocking matrix, generalized sidelobe canceller, interference suppression, matrix algebra, noise suppression, speech enhancement, transfer function estimation, transfer functions}},
  pages        = {{73--76}},
  title        = {{{Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller}}},
  doi          = {{10.1109/ICASSP.2008.4517549}},
  year         = {{2008}},
}

@inproceedings{11939,
  abstract     = {{In this paper a switching linear dynamical model (SLDM) approach for speech feature enhancement is improved by employing more accurate models for the dynamics of speech and noise. The model of the clean speech feature trajectory is improved by augmenting the state vector to capture information derived from the delta features. Further a hidden noise state variable is introduced to obtain a more elaborated model for the noise dynamics. Approximate Bayesian inference in the SLDM is carried out by a bank of extended Kalman filters, whose outputs are combined according to the a posteriori probability of the individual state models. Experimental results on the AURORA2 database show improved recognition accuracy.}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}},
  keywords     = {{a posteriori probability, AURORA2 database, Bayesian inference, Bayes methods, channel bank filters, extended Kalman filter banks, hidden noise state variable, Kalman filters, noise dynamics, speech enhancement, speech feature enhancement, speech feature trajectory, switching linear dynamical model approach}},
  pages        = {{4409--4412}},
  title        = {{{Modeling the dynamics of speech and noise for speech feature enhancement in ASR}}},
  doi          = {{10.1109/ICASSP.2008.4518633}},
  year         = {{2008}},
}

@inproceedings{11943,
  abstract     = {{A marginalized particle filter is proposed for performing single channel speech enhancement with a non-linear dynamic state model. The system consists of a particle filter for tracking line spectral pair (LSP) parameters and a Kalman filter per particle for speech enhancement. The state model for the LSPs has been learnt on clean speech training data. In our approach parameters and speech samples are processed at different time scales by assuming the parameters to be constant for small blocks of data. Further enhancement is obtained by an iteration which can be applied on these small blocks. The experiments show that similar SNR gains are obtained as with the Kalman-LM-iterative algorithm. However better values of the noise level and the log-spectral distance are achieved}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)}},
  keywords     = {{clean speech training data, iterative methods, iterative speech enhancement, Kalman filter, Kalman filters, Kalman-LM-iterative algorithm, line spectral pair parameters, log-spectral distance, marginalized particle filter, noise level, nonlinear dynamic state speech model, particle filtering (numerical methods), single channel speech enhancement, SNR gains, speech enhancement, speech samples}},
  pages        = {{I}},
  title        = {{{Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters}}},
  doi          = {{10.1109/ICASSP.2006.1660058}},
  volume       = {{1}},
  year         = {{2006}},
}

@inproceedings{11931,
  abstract     = {{The paper is concerned with binaural signal processing for a bimodal human-robot interface with hearing and vision. The two microphone signals are processed to obtain an enhanced single-channel input signal for the subsequent speech recognizer and to localize the acoustic source, an important information for establishing a natural human-robot communication. We utilize a robust adaptive algorithm for filter-and-sum beamforming (FSB) and extract speaker direction information from the resulting FIR filter coefficients. Further, particle filtering is applied which conducts a nonlinear Bayesian tracking of speaker movement. Good location accuracy can be achieved even in highly reverberant environments. The results obtained outperform the conventional generalized cross correlation (GCC) method.}},
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE Workshop on Multimedia Signal Processing (MMSP 2004)}},
  keywords     = {{bimodal human-robot interface, binaural signal processing, enhanced single-channel input signal, filter-and-sum beamforming, filtering theory, FIR filter coefficient, generalized cross correlation method, microphones, microphone signal, nonlinear Bayesian tracking, particle filtering, robust adaptive algorithm, robust speaker direction estimation, signal processing, speech enhancement, speech recognition, speech recognizer, user interfaces}},
  pages        = {{367--370}},
  title        = {{{Robust speaker direction estimation with particle filtering}}},
  doi          = {{10.1109/MMSP.2004.1436569}},
  year         = {{2004}},
}

