@inproceedings{11806,
  abstract     = {{Microphone arrays represent the basis for many challenging acoustic sensing tasks. The accuracy of techniques like beamforming directly depends on a precise knowledge of the relative positions of the sensors used. Unfortunately, for certain use cases manually measuring the geometry of an array is not feasible due to practical constraints. In this paper we present an approach to unsupervised shape calibration of microphone array networks. We developed a hierarchical procedure that first performs local shape calibration based on coherence analysis and then employs SRP-PHAT in a network calibration method. Practical experiments demonstrate the effectiveness of our approach especially for highly reverberant acoustic environments.}},
  author       = {{Hennecke, Marius and Ploetz, Thomas and Fink, Gernot A. and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE/SP 15th Workshop on Statistical Signal Processing (SSP 2009)}},
  keywords     = {{acoustic sensing tasks, array geometry, calibration, coherence analysis, hierarchical procedure, local shape calibration, microphone array networks, microphone arrays, network calibration method, sensor arrays, SRP-PHAT, unsupervised shape calibration}},
  pages        = {{257--260}},
  title        = {{{A hierarchical approach to unsupervised shape calibration of microphone array networks}}},
  doi          = {{10.1109/SSP.2009.5278589}},
  year         = {{2009}},
}

@inproceedings{11899,
  abstract     = {{In this paper we present a system for identifying and localizingspeakers using distant microphone arrays and a steerablepan-tilt-zoom camera. Audio and video streams are processedin real-time to obtain the diarization information {grqq}who speakswhen and where'' with low latency to be used in advanced videoconferencing systems or user-adaptive interfaces. A key featureof the proposed system is to first glean information about thespeaker{\rq}s location and identity from the audio and visual datastreams separately and then to fuse these data in a probabilisticframework employing the Viterbi algorithm. Here, visual evidenceof a person is utilized through a priori state probabilities,while location and speaker change information are employedvia time-variant transition probablities. Experiments show thatvideo information yields a substantial improvement comparedto pure audio-based diarization.}},
  author       = {{Schmalenstroeer, Joerg and Kelling, Martin and Leutnant, Volker and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2009}},
  title        = {{{Fusing Audio and Video Information for Online Speaker Diarization}}},
  year         = {{2009}},
}

@article{11776,
  abstract     = {{The term uncertainty decoding has been phrased for a class of robustness enhancing algorithms in automatic speech recognition that replace point estimates and plug-in rules by posterior densities and optimal decision rules. While uncertainty can be incorporated in the model domain, in the feature domain, or even in both, we concentrate here on feature domain approaches as they tend to be computationally less demanding. We derive optimal decision rules in the presence of uncertain observations and discuss simplifications which result in computationally efficient realizations. The usefulness of the presented statistical framework is then exemplified for two types of realworld problems: The first is improving the robustness of speech recognition towards incomplete or corrupted feature vectors due to a lossy communication link between the speech capturing front end and the backend recognition engine. And the second is the well-known and extensively studied issue of improving the robustness of the recognizer towards environmental noise.}},
  author       = {{Haeb-Umbach, Reinhold}},
  journal      = {{2008 ITG Conference on Voice Communication (SprachKommunikation)}},
  pages        = {{1--7}},
  title        = {{{Uncertainty Decoding in Automatic Speech Recognition}}},
  year         = {{2008}},
}

@inbook{11789,
  abstract     = {{In distributed and network speech recognition the actual recognition task is not carried out on the user{\rq}s terminal but rather on a remote server in the network. While there are good reasons for doing so, a disadvantage of this client-server architecture is clearly that the communication medium may introduce errors, which then impairs speech recognition accuracy. Even sophisticated channel coding cannot completely prevent the occurrence of residual bit errors in the case of temporarily adverse channel conditions, and in packet-oriented transmission packets of data may arrive too late for the given real-time constraints and have to be declared lost. The goal of error concealment is to reduce the detrimental effect that such errors may induce on the recipient of the transmitted speech signal by exploiting residual redundancy in the bit stream at the source coder output. In classical speech transmission a human is the recipient, and erroneous data are reconstructed so as to reduce the subjectively annoying effect of corrupted bits or lost packets. Here, however, a statistical classifier is at the receiving end, which can benefit from knowledge about the quality of the reconstruction. In this book chapter we show how the classical Bayesian decision rule needs to be modified to account for uncertain features, and illustrate how the required feature posterior density can be estimated in the case of distributed speech recognition. Some other techniques for error concealment can be related to this approach. Experimental results are given for both a small and a medium vocabulary recognition task and both for a channel exhibiting bit errors and a packet erasure channel.}},
  author       = {{Haeb-Umbach, Reinhold and Ion, Valentin}},
  booktitle    = {{Automatic Speech Recognition on Mobile Devices and over Communication Networks}},
  editor       = {{Lindenberg, Borge and Tan, Zheng-Hua}},
  pages        = {{187--210}},
  publisher    = {{Springer}},
  title        = {{{Error Concealement}}},
  volume       = {{Advances in Computer Vision and Pattern Recognition}},
  year         = {{2008}},
}

@article{11820,
  abstract     = {{In this paper, we derive an uncertainty decoding rule for automatic speech recognition (ASR), which accounts for both corrupted observations and inter-frame correlation. The conditional independence assumption, prevalent in hidden Markov model-based ASR, is relaxed to obtain a clean speech posterior that is conditioned on the complete observed feature vector sequence. This is a more informative posterior than one conditioned only on the current observation. The novel decoding is used to obtain a transmission-error robust remote ASR system, where the speech capturing unit is connected to the decoder via an error-prone communication network. We show how the clean speech posterior can be computed for communication links being characterized by either bit errors or packet loss. Recognition results are presented for both distributed and network speech recognition, where in the latter case common voice-over-IP codecs are employed.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{automatic speech recognition, bit errors, codecs, communication links, corrupted observations, decoding, distributed speech recognition, error-prone communication network, feature vector sequence, hidden Markov model-based ASR, hidden Markov models, inter-frame correlation, Internet telephony, network speech recognition, packet loss, speech posterior, speech recognition, transmission error robust speech recognition, uncertainty decoding, voice-over-IP codecs}},
  number       = {{5}},
  pages        = {{1047--1060}},
  title        = {{{A Novel Uncertainty Decoding Rule With Applications to Transmission Error Robust Speech Recognition}}},
  doi          = {{10.1109/TASL.2008.925879}},
  volume       = {{16}},
  year         = {{2008}},
}

@article{11821,
  abstract     = {{This paper addresses the robustness of automatic speech recognition to environmental noise. In order to account for reliability of the clean feature estimate we employ the feature posterior density conditioned on observed noisy features to perform uncertainty decoding. We investigate two approaches to estimate the posterior using a discrete feature space, first conditioning only on the current observation, and second on the whole feature sequence of an utterance. Experiments with Aurora 2 showed that the latter provides slightly better performance, as it allows for exploiting the temporal correlations between consecutive features.}},
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  journal      = {{2008 ITG Conference on Voice Communication (SprachKommunikation)}},
  pages        = {{1--4}},
  title        = {{{Investigations into Uncertainty Decoding Employing a Discrete Feature Space for Noise Robust Automatic Speech Recognition}}},
  year         = {{2008}},
}

@inproceedings{11851,
  abstract     = {{In diesem Beitrag werden zwei neuartige akustische Strahlformungsalgorithmen fuer einen Einsatz im KFZ diskutiert. In beiden Verfahren ist fuer jede Frequenzkomponente der Eigenvektor zum groessten Eigenwert eines verallgemeinerten Eigenwertproblems zu bestimmen: bei der einen Varianate stellen die Filterkoeffizienten gerade obigen Eigenvektor dar. Sprachverzerrungen werden mit einem einkanaligen Nachfilter kompensiert. In der anderen Variante, welche die Struktur eines ''Generalized Sidelobe Cancellers'' (GSC) hat, basiert die Blockiermatrix auf obiger Eigenvektorzerlegung.}},
  author       = {{Krueger, Alexander and Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  booktitle    = {{34. Deutsche Jahrestagung fuer Akustik (DAGA 2008)}},
  title        = {{{Blinde Akustische Strahlformung fuer Anwendungen im KFZ}}},
  year         = {{2008}},
}

@article{11914,
  abstract     = {{This paper considers the convolutive blind source separation of speech sources in the presence of spatially correlated noise. We introduce a method for estimating the scaled mixing matrix from the sources to the microphones even if coherent noise is present. This is achieved by combining time-frequency sparseness with the generalized eigenvalue decomposition of the power spectral density matrix (PSD) of the noisy speech and noise-only microphone signals. Separation is performed by spatial filtering with coefficients constructed by Gram-Schmidt orthogonalization which places spatial nulls at the interferer{^A}?`s direction. Experimental results show that our approach is capable of separating 2 sources in a reverberant environment (RT60=0ms..500ms) degraded by significant directional noise.}},
  author       = {{Tran Vu, Dang Hai and Haeb-Umbach, Reinhold}},
  journal      = {{2008 ITG Conference on Voice Communication (SprachKommunikation)}},
  pages        = {{1--4}},
  title        = {{{Blind Speech Separation in Presence of Correlated Noise with Generalized Eigenvector Beamforming}}},
  year         = {{2008}},
}

@inproceedings{11915,
  abstract     = {{This Paper deals with a new Technique for multi-channel separation of speech signals from convolutive mixtures under coherent noise. We demonstrate how the scaled transfer functions from the sources to the microphones can be estimated even in the presence of stationary coherent noise. The key to this are generalized eigenvalue decompositions of the power spectral density (PSD) matrices of the noisy speech and noise-only microphone signals with a controlled estimation of these matrices exploiting time-frequency sparseness of the speech sources. Separation is further improved by subsequent Gram-Schmidt orthogonalization which places spatial nulls at the interferers{\rq} directions, while noise reduction is improved by employing a novel blocking matrix and adaptive interference canceller in a Generalized Sidelobe Canceller (GSC)-like structure. We report promising experimental results for 2 speech sources with significant coherent noise in reverberant environments (RT60=0oms..500ms).}},
  author       = {{Tran Vu, Dang Hai and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{International Workshop on Acoustic Echo and Noise Control (IWAENC 2008)}},
  title        = {{{Generalized Eigenvector Blind Speech Separation Under Coherent Noise In A GSC Configuration}}},
  year         = {{2008}},
}

@inproceedings{11935,
  abstract     = {{The generalized sidelobe canceller by Griffith and Jim is a robust beamforming method to enhance a desired (speech) signal in the presence of stationary noise. Its performance depends to a high degree on the construction of the blocking matrix which produces noise reference signals for the subsequent adaptive interference canceller. Especially in reverberated environments the beamformer may suffer from signal leakage and reduced noise suppression. In this paper a new blocking matrix is proposed. It is based on a generalized eigenvalue problem whose solution provides an indirect estimation of the transfer functions from the source to the sensors. The quality of the new generalized eigenvector blocking matrix is studied in simulated rooms with different reverberation times and is compared to alternatives proposed in the literature.}},
  author       = {{Warsitz, Ernst and Krueger, Alexander and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}},
  keywords     = {{adaptive interference canceller, adaptive signal processing, array signal processing, beamforming method, eigenvalues and eigenfunctions, generalized eigenvector blocking matrix, generalized sidelobe canceller, interference suppression, matrix algebra, noise suppression, speech enhancement, transfer function estimation, transfer functions}},
  pages        = {{73--76}},
  title        = {{{Speech enhancement with a new generalized eigenvector blocking matrix for application in a generalized sidelobe canceller}}},
  doi          = {{10.1109/ICASSP.2008.4517549}},
  year         = {{2008}},
}

@inproceedings{11939,
  abstract     = {{In this paper a switching linear dynamical model (SLDM) approach for speech feature enhancement is improved by employing more accurate models for the dynamics of speech and noise. The model of the clean speech feature trajectory is improved by augmenting the state vector to capture information derived from the delta features. Further a hidden noise state variable is introduced to obtain a more elaborated model for the noise dynamics. Approximate Bayesian inference in the SLDM is carried out by a bank of extended Kalman filters, whose outputs are combined according to the a posteriori probability of the individual state models. Experimental results on the AURORA2 database show improved recognition accuracy.}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2008)}},
  keywords     = {{a posteriori probability, AURORA2 database, Bayesian inference, Bayes methods, channel bank filters, extended Kalman filter banks, hidden noise state variable, Kalman filters, noise dynamics, speech enhancement, speech feature enhancement, speech feature trajectory, switching linear dynamical model approach}},
  pages        = {{4409--4412}},
  title        = {{{Modeling the dynamics of speech and noise for speech feature enhancement in ASR}}},
  doi          = {{10.1109/ICASSP.2008.4518633}},
  year         = {{2008}},
}

@article{11940,
  abstract     = {{In this paper, the noise estimation for model-based speech feature enhancement in automatic speech recognition (ASR) is investigated. Beside a stationary noise prior, three linear state space models for the (cepstral) noise process are considered. We have derived novel EM algorithms for the estimation of the noise model parameters: A blockwise EM algorithm is applied on noise-only input data. It is supposed to be used during the offline training mode of the recognizer. Further a sequential online EM algorithm is employed to adapt the observation variance in recognition mode which works as well under the asumption of a stationary noise prior and a linear state model for the noise. Experiments on the AURORA4 database lead to improved recognition results with the new state model compared to the assumption of stationary noise.}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold}},
  journal      = {{2008 ITG Conference on Voice Communication (SprachKommunikation)}},
  pages        = {{1--4}},
  title        = {{{A novel approach to noise estimation in model-based speech feature enhancement}}},
  year         = {{2008}},
}

@article{11944,
  abstract     = {{In this paper, a novel segmental Hidden Markov Model (HMM) is proposed. The model is based on a modified emission density where additional statistical dependencies between subsequent frames of the speech signal are considered. In the following we derive an effective search strategy for the modified statistical model. Further an approach to parameter reduction is introduced. Experiments were carried out on the AURORA2 database where consistent im}},
  author       = {{Windmann, Stefan and Haeb-Umbach, Reinhold and Leutnant, Volker}},
  journal      = {{2008 ITG Conference on Voice Communication (SprachKommunikation)}},
  pages        = {{1--4}},
  title        = {{{A segmental HMM based on a modified emission probability}}},
  year         = {{2008}},
}

@inproceedings{11720,
  author       = {{Bevermeier, Maik and Ebel, Tobias and Haeb-Umbach, Reinhold}},
  booktitle    = {{Multi-Carrier Spread Spectrum 2007}},
  title        = {{{Channel Estimation by Exploiting Sublayer Information in OFDM Systems}}},
  year         = {{2007}},
}

@inproceedings{11722,
  author       = {{Bevermeier, Maik and Haeb-Umbach, Reinhold}},
  booktitle    = {{Multi-Carrier Spread Spectrum 2007}},
  title        = {{{Combined Time and Frequency Domain OFDM Channel Estimation}}},
  year         = {{2007}},
}

@inproceedings{11785,
  abstract     = {{In this paper we present a novel channel impulse response estimation technique for block-oriented OFDM transmission based on combining estimators: the estimates provided by a Kalman filter operating in the time domain and a Wiener filter in the frequency domain are optimally combined by taking into account their estimated error covariances. The resulting estimator turns out to be identical to the MAP estimator of correlated jointly Gaussian mean vectors. Different variants of the proposed scheme are experimentally investigated in an EEEE 802.11a-like system setup. They compare favourably with known approaches from the literature resulting in reduced mean square estimation error and bit error rate. Further, robustness and complexity issues are discussed}},
  author       = {{Haeb-Umbach, Reinhold and Bevermeier, Maik}},
  booktitle    = {{IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2007)}},
  keywords     = {{bit error rate, block-oriented OFDM transmission, channel estimation, channel impulse response estimation, combining estimators, error statistics, frequency domain estimation, Gaussian mean vectors, Gaussian processes, Kalman filter, Kalman filters, MAP estimator, maximum likelihood estimation, OFDM channel estimation, OFDM modulation, time domain estimation, time-frequency analysis, Wiener filter, Wiener filters}},
  pages        = {{III--277--III--280}},
  title        = {{{OFDM Channel Estimation Based on Combined Estimation in Time and Frequency Domain}}},
  doi          = {{10.1109/ICASSP.2007.366526}},
  volume       = {{3}},
  year         = {{2007}},
}

@article{11799,
  abstract     = {{In this paper, we propose a novel similarity measure to be used for localizing mobile terminals by comparing measured signal power levels with a database of predictions. The proposed measure provides the possibility to incorporate inherent information about signal power level measurements requested by the serving base station but not reported by the mobile terminal. Increased positioning accuracy was observed both in simulations and with real field data}},
  author       = {{Haeb-Umbach, Reinhold and Peschke, Sven}},
  journal      = {{IEEE Transactions on Vehicular Technology}},
  keywords     = {{cellular phone positioning, cellular radio, measured signal power levels, mobile handsets, mobility management (mobile radio)}},
  number       = {{1}},
  pages        = {{368--372}},
  title        = {{{A Novel Similarity Measure for Positioning Cellular Phones by a Comparison With a Database of Signal Power Levels}}},
  doi          = {{10.1109/TVT.2006.889563}},
  volume       = {{56}},
  year         = {{2007}},
}

@inproceedings{11822,
  author       = {{Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{Interspeech 2007}},
  title        = {{{Multi-Resolution Soft Features for Channel-Robust Distributed Speech Recognition}}},
  year         = {{2007}},
}

@inproceedings{11883,
  abstract     = {{In this paper, we experimentally evaluate algorithms for velocity estimation of a GSM 900 mobile terminal which are based on the analysis of the statistical properties of the fast fading process. It is shown how theses statistics can be obtained from the training sequences present in downlink transmission bursts without establishing an active connection. Realistic simulations of a GSM channel according to the COST 207 channel models have been conducted. These models incorporate effects like multipath propagation, fading, cochannel interference and additive noise. It is shown that velocity estimation by searching for the maximum slope of the power density spectrum of the fast fading performs best.}},
  author       = {{Peschke, Sven and Haeb-Umbach, Reinhold}},
  booktitle    = {{4th Workshop on Positioning Navigation and Communication (WPNC 2007)}},
  keywords     = {{additive noise, cellular radio, channel estimation, cochannel interference, COST 207 channel models, downlink transmission bursts, fading channels, fading process, GSM downlink signalling, mobile terminals, multipath channels, multipath propagation, power density spectrum, statistical analysis, statistical properties, telecommunication links, telecommunication terminals, velocity estimation}},
  pages        = {{217--222}},
  title        = {{{Velocity Estimation of Mobile Terminals by Exploiting GSM Downlink Signalling}}},
  doi          = {{10.1109/WPNC.2007.353637}},
  year         = {{2007}},
}

@article{11927,
  abstract     = {{Maximizing the output signal-to-noise ratio (SNR) of a sensor array in the presence of spatially colored noise leads to a generalized eigenvalue problem. While this approach has extensively been employed in narrowband (antenna) array beamforming, it is typically not used for broadband (microphone) array beamforming due to the uncontrolled amount of speech distortion introduced by a narrowband SNR criterion. In this paper, we show how the distortion of the desired signal can be controlled by a single-channel post-filter, resulting in a performance comparable to the generalized minimum variance distortionless response beamformer, where arbitrary transfer functions relate the source and the microphones. Results are given both for directional and diffuse noise. A novel gradient ascent adaptation algorithm is presented, and its good convergence properties are experimentally revealed by comparison with alternatives from the literature. A key feature of the proposed beamformer is that it operates blindly, i.e., it neither requires knowledge about the array geometry nor an explicit estimation of the transfer functions from source to sensors or the direction-of-arrival.}},
  author       = {{Warsitz, Ernst and Haeb-Umbach, Reinhold}},
  journal      = {{IEEE Transactions on Audio, Speech, and Language Processing}},
  keywords     = {{acoustic signal processing, arbitrary transfer function, array signal processing, blind acoustic beamforming, direction-of-arrival, direction-of-arrival estimation, eigenvalues and eigenfunctions, generalized eigenvalue decomposition, gradient ascent adaptation algorithm, microphone arrays, microphones, narrowband array beamforming, sensor array, single-channel post-filter, spatially colored noise, transfer functions}},
  number       = {{5}},
  pages        = {{1529--1539}},
  title        = {{{Blind Acoustic Beamforming Based on Generalized Eigenvalue Decomposition}}},
  doi          = {{10.1109/TASL.2007.898454}},
  volume       = {{15}},
  year         = {{2007}},
}

