@inproceedings{11934,
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold and Tran Vu, Dang Hai}},
  booktitle    = {{Interspeech 2007}},
  title        = {{{Blind Adaptive Principal Eigenvector Beamforming for Acoustical Source Separation}}},
  year         = {{2007}},
}

@inproceedings{11941,
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2007}},
  title        = {{{An Approach to Iterative Speech Feature Enhancement and Recognition}}},
  year         = {{2007}},
}

@inproceedings{11893,
  author       = {{Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2007}},
  title        = {{{Joint Speaker Segmentation, Localization and Identification for Streaming Audio}}},
  year         = {{2007}},
}

@inproceedings{11901,
  author       = {{Schmalenstroeer, Joerg and Leutnant, Volker and Haeb-Umbach, Reinhold}},
  booktitle    = {{AMI-07 - European Conference on Ambient Intelligence}},
  title        = {{{Amigo Context Management Service with Applications in Ambient Communication Scenarios}}},
  year         = {{2007}},
}

@inproceedings{11933,
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold and Schmalenstroeer, Joerg}},
  booktitle    = {{33. Deutsche Jahrestagung fuer Akustik (DAGA 2007)}},
  title        = {{{Zweistufige Sprache/Pause-Detektion in stark gestoerter Umgebung}}},
  year         = {{2007}},
}

@inproceedings{11902,
  author       = {{Schmalenstroeer, Joerg and Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  booktitle    = {{33. Deutsche Jahrestagung fuer Akustik (DAGA 2007)}},
  title        = {{{Projekt Amigo - Sprachsignalverarbeitung im vernetzten Haus}}},
  year         = {{2007}},
}

@inproceedings{11823,
  abstract     = {{In this study we evaluate transmission error compensation techniques for distributed speech recognition systems based on modification of the speech decoder. The candidates are marginalization, weighted Viterbi and our recently proposed soft-feature uncertainty decoding. For the latter, it is shown how the Bayesian speech recognition approach must be reformulated for recognition at the server side. The resulting predictive classifier is able to take account of the transmission errors by changing the contribution of the affected speech features to the acoustic score. The comparison of the experimental results has proven the superiority of our approach.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{7. ITG-Fachtagung Sprachkommunikation}},
  title        = {{{Comparison of Decoder-based Transmission Error Compensation Techniques for Distributed Speech Recognition}}},
  year         = {{2006}},
}

@inproceedings{11824,
  abstract     = {{Soft-feature based speech recognition, which is an example of uncertainty decoding, has been proven to be a robust error mitigation method for distributed speech recognition over wireless channels exhibiting bit errors. In this paper we extend this concept to packet-oriented transmissions. The a posteriori probability density function of the lost feature vector, given the closest received neighbours, is computed. In the experiments, the nearest frame repetition, which is shown to be equivalent to the MAP estimate, outperforms the MMSE estimate for long bursts. Taking the variance into account at the speech recognition stage results in superior performance compared to classical schemes using point estimates. A computationally and memory efficient implementation of the proposed packet loss compensation scheme based on table lookup is presented}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)}},
  keywords     = {{distributed speech recognition, least mean squares methods, MAP estimate, maximum likelihood estimation, MMSE estimate, packet loss compensation scheme, packet switched communication, posteriori probability density function, robust error mitigation method, soft-features, speech recognition, table lookup, voice communication, wireless channels}},
  pages        = {{I}},
  title        = {{{An Inexpensive Packet Loss Compensation Scheme for Distributed Speech Recognition Based on Soft-Features}}},
  doi          = {{10.1109/ICASSP.2006.1659984}},
  volume       = {{1}},
  year         = {{2006}},
}

@article{11825,
  abstract     = {{In this paper, we propose an enhanced error concealment strategy at the server side of a distributed speech recognition (DSR) system, which is fully compatible with the existing DSR standard. It is based on a Bayesian approach, where the a posteriori probability density of the error-free feature vector is computed, given all received feature vectors which are possibly corrupted by transmission errors. Rather than computing a point estimate, such as the MMSE estimate, and plugging it into the Bayesian decision rule, we employ uncertainty decoding, which results in an integration over the uncertainty in the feature domain. In a typical scenario the communication between the thin client, often a mobile device, and the recognition server spreads across heterogeneous networks. Both bit errors on circuit-switched links and lost data packets on IP connections are mitigated by our approach in a unified manner. The experiments reveal improved robustness both for small- and large-vocabulary recognition tasks.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  journal      = {{Speech Communication}},
  keywords     = {{Channel error robustness, Distributed speech recognition, Soft features, Uncertainty decoding}},
  number       = {{11}},
  pages        = {{1435--1446}},
  title        = {{{Uncertainty decoding for distributed speech recognition over error-prone networks}}},
  doi          = {{10.1016/j.specom.2006.03.007}},
  volume       = {{48}},
  year         = {{2006}},
}

@inproceedings{11826,
  abstract     = {{The accuracy of distributed speech recognition has been shown to be very sensitive to errors occurring during transmission. One reason for this is that the classifier, usually trained under error free conditions, is unable to cope with the mismatch between an error free and error prone channel. In this paper we present a novel decision rule for classification which is able to account for channel errors. To achieve this, the classical Bayesian speech recognition approach has been reformulated for the server side, where the observation is known only to the extent, as is given by its a posteriori density function. We present a method to estimate the a posteriori density which is based on a Markov model of the source, which captures correlations of both static and dynamic features. A practical implementation is given, accompanied by experimental results for distributed speech recognition over an IP-network.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2006}},
  title        = {{{Improved Source Modeling and Predictive Classification for Channel Robust Speech Recognition}}},
  year         = {{2006}},
}

@inproceedings{11884,
  abstract     = {{In this paper we present the design of a particle filter for post filtering instantaneous positioning estimates of GSM mobile terminals. The instantaneous estimates are obtained by comparing signal power levels, which are reported by the mobile terminal to the base station, with a database of predictions using a novel statistically motivated similarity measure. Unlike a simple Euclidian distance measure, the proposed scheme incorporates inherent information about signal power level measurements requested by the serving base station but not reported by the mobile terminal. Furthermore, we show how the Monte Carlo method of particle filtering helps to obtain better position estimates and, surprisingly, also helps to reduce the computational complexity. Results are presented for real field data.}},
  author       = {{Peschke, Sven and Haeb-Umbach, Reinhold}},
  booktitle    = {{European Navigation Conference \& Exhibition (ENC 2006)}},
  title        = {{{A Probabilistic Similarity Measure and a Non-Linear Post-Filter for Mobile Phone Positioning using GSM Signal Power Measurements}}},
  year         = {{2006}},
}

@inproceedings{11885,
  abstract     = {{In this paper we present a novel and statistically motivated similarity measure for database assisted positioning of GSM mobile terminals by evaluating signal power level reports which are transmitted regulary. Unlike a simple Euclidian distance measure, the proposed scheme incorporates inherent information about signal power level measurements requested by the serving base station but not reported by the mobile terminal. Furthermore we show how the Monte Carlo method of nonlinear post filtering using particle filtering helps to obtain better position estimates and surprisingly also helps to reduce the computational complexity. Results are presented for real field data.}},
  author       = {{Peschke, Sven and Haeb-Umbach, Reinhold}},
  booktitle    = {{3rd Workshop on Positioning Navigation and Communication (WPNC 2006)}},
  title        = {{{Particle Filtering of Database assisted Positioning Estimates using a novel Similarity Measure for GSM Signal Power Level Measurements}}},
  year         = {{2006}},
}

@inproceedings{11928,
  abstract     = {{Broadband adaptive beamformers, which use a narrowband SNR-maximization optimization criterion for noise reduction, typically cause distortions of the desired speech signal at the beamformer output. In this paper two methods are investigated to control the speech distortion by comparing the eigenvector beamformer with a maximum likelihood beamformer: One is an analytic solution for the ideal case of absence of reverberation and the other one is a statistically motivated approach. We use the recently introduced gradient-ascent algorithm for adaptive principal eigenvector beamforming and then normalize the filter coefficients by the proposed distortion control methods. Experimental results in terms of the achievable SNR gain and a perceptual speech quality measure are given for the normalized eigenvector beamformer and are compared to standard beamforming methods.}},
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  booktitle    = {{32. Deutsche Jahrestagung fuer Akustik (DAGA 2006)}},
  title        = {{{Mehrkanalige Sprachsignalverarbeitung durch adaptives Eigenbeamforming fuer Freisprecheinrichtungen im Kraftfahrzeug}}},
  year         = {{2006}},
}

@inproceedings{11929,
  abstract     = {{Broadband adaptive beamformers, which use a narrowband SNR-maximization optimization criterion for noise reduction, typically cause distortions of the desired speech signal at the beamformer output. In this paper two methodsare investigated to control the speech distortion by comparing the eigenvector beamformer with a maximum likelihood beamformer: One is an analytic solution for the ideal case of absence of reverberation and the other one is a statistically motivated approach. We use the recently introduced gradient-ascent algorithm for adaptive principal eigenvector beamforming and then normalize the filter coefficient s by the proposed distortion control methods. Experimental results in terms of the achievable SNR gain and a perceptual speech quality measure are given for the normalized eigenvector beamformer and are compared to standard beamforming methods.}},
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  booktitle    = {{International Workshop on Acoustic Echo and Noise Control (IWAENC 2006)}},
  title        = {{{Controlling Speech Distortion in Adaptive Frequency-Domain Principal Eigenvector Beamforming}}},
  year         = {{2006}},
}

@inproceedings{11942,
  abstract     = {{Es wird ein marginalisiertes Partikelfilter beschrieben, das zur einkanaligen Sprachsignalverbesserung mit einem nichtlinearen dynamischen Zustandsmodell eingesetzt werden soll. Das System besteht aus einem Partikelfilter zum Tracking von LSP-Parametern und einem Kalman-Filter fuer jedes Partikel, das zur Sprachsignalverbesserung verwendet wird. In unserem Ansatz wird angenommen, dass die Parameter in kurzen Sprachsignalbloecken konstant sind, waehrend das Sprachsignal sich mit jedem Abtastwert aendert. Bei weissem Rauschen werden aehnliche SNR-Gewinne wie mit einem Kalman-EM-iterative Algorithmus erzielt, waehrend das Hintergrundrauschen und die Log-spektrale Distanz etwas geringer sind. Mit einem erweiterten Zustandsmodell wurden auch Untersuchungen fuer farbiges Rauschen durchgefuehrt.}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{7. ITG-Fachtagung Sprachkommunikation}},
  title        = {{{Einkanalige Sprachsignalverbesserung mit Hilfe eines marginalisierten Partikelfilters}}},
  year         = {{2006}},
}

@inproceedings{11943,
  abstract     = {{A marginalized particle filter is proposed for performing single channel speech enhancement with a non-linear dynamic state model. The system consists of a particle filter for tracking line spectral pair (LSP) parameters and a Kalman filter per particle for speech enhancement. The state model for the LSPs has been learnt on clean speech training data. In our approach parameters and speech samples are processed at different time scales by assuming the parameters to be constant for small blocks of data. Further enhancement is obtained by an iteration which can be applied on these small blocks. The experiments show that similar SNR gains are obtained as with the Kalman-LM-iterative algorithm. However better values of the noise level and the log-spectral distance are achieved}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2006)}},
  keywords     = {{clean speech training data, iterative methods, iterative speech enhancement, Kalman filter, Kalman filters, Kalman-LM-iterative algorithm, line spectral pair parameters, log-spectral distance, marginalized particle filter, noise level, nonlinear dynamic state speech model, particle filtering (numerical methods), single channel speech enhancement, SNR gains, speech enhancement, speech samples}},
  pages        = {{I}},
  title        = {{{Iterative Speech Enhancement using a Non-Linear Dynamic State Model of Speech and its Parameters}}},
  doi          = {{10.1109/ICASSP.2006.1660058}},
  volume       = {{1}},
  year         = {{2006}},
}

@inproceedings{11894,
  abstract     = {{In this paper we consider the problem of detecting speaker changes in audio signals recorded by distant microphones. It is shown that the possibility to exploit the spatial separation of speakers more than makes up the degradation in detection accuracy due to the increased source-to-sensor distance compared to close-talking microphones. Speaker direction information is derived from the filter coefficients of an adaptive Filter-and-Sum Beamformer and is combined with BIC analysis. The experimental results reveal significant improvements compared to BIC-only change detection, be it with the distant or close-talking microphone.}},
  author       = {{Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2006}},
  title        = {{{Online Speaker Change Detection by Combining BIC with Microphone Array Beamforming}}},
  year         = {{2006}},
}

@inproceedings{11803,
  abstract     = {{In this paper we propose a novel adaptation algorithm for Filter-and-Sum beamforming in spatially correlated noise. Deterministic and stochastic gradient ascent algorithms are derived from a constrained optimization problem, which iteratively estimate the principal eigenvecto r of a generalized eigenvalue problem. The method does not require an explicit estimation of the speaker location. It is shown that the well-known Delay-and-Sum beamformer and the previously introduced Filter-and-Sum beamformer in spatially white noise are obtained as special cases. Further, bounds on the maximally achievable SNR gains are derived and it is shown that the proposed adaptation algorithm is able to approach these performance bounds.}},
  author       = {{Haeb-Umbach, Reinhold and Warsitz, Ernst}},
  booktitle    = {{International Workshop on Acoustic Echo and Noise Control (IWAENC 2005)}},
  title        = {{{Adaptive Filter-and-Sum Beamforming in Spatially Correlated Noise}}},
  year         = {{2005}},
}

@inproceedings{11827,
  abstract     = {{The transmission errors in a wireless or packet oriented network may dramatically decrease the performance of a distributed speech recognition DSR) system. Error concealment has been shown to be an effective way to mantain an acceptable word error rate when dealing with error prone communication channels. In this paper we propose an extension of our previously introduced soft features approach for the case that the soft-output of the channel decoder is not available at the server side of the DSR system. We found a simple method to estimate bit reliability information which still gives good speech recognition results. It is shown that some other error concealment schemes turn out to be special cases of the method proposed here.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2005}},
  title        = {{{A Unified Probabilistic Approach to Error Concealment for Distributed Speech Recognition}}},
  year         = {{2005}},
}

@inproceedings{11828,
  abstract     = {{In this paper we present a comparison of the recently proposed Soft-Feature Distributed Speech Recognition (SFDSR) with the two evaluated candidate codecs for Speech Enabled Services over wireless networks: Adaptive Multirate Codec (AMR) and the ETSI Extended Advanced Front-End for Distributed Speech Recognition (XAFE). It is shown that SFDSR achieves the best recognition performance on a simulated GSM transmission, followed by XAFE and AMR.We also present some new results concerning SFDSR which demonstrate the versatility of the approach. Further, a simple method is introduced which considerably reduces the computational effort.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2005)}},
  keywords     = {{adaptive codes, adaptive multirate codec, AMR, distributed speech recognition, ETSI, extended advanced front-end, recognition performance, SFDSR, simulated GSM transmission, soft-feature distributed speech recognition, speech codecs, speech coding, speech recognition, variable rate codes, XAFE}},
  pages        = {{333--336}},
  title        = {{{A Comparison of Soft-Feature Distributed Speech Recognition with Candidate Codecs for Speech Enabled Mobile Services}}},
  doi          = {{10.1109/ICASSP.2005.1415118}},
  volume       = {{1}},
  year         = {{2005}},
}

