@inproceedings{33806,
  author       = {{Afifi, Haitham and Karl, Holger and Gburrek, Tobias and Schmalenstroeer, Joerg}},
  booktitle    = {{2022 International Wireless Communications and Mobile Computing (IWCMC)}},
  publisher    = {{IEEE}},
  title        = {{{Data-driven Time Synchronization in Wireless Multimedia Networks}}},
  doi          = {{10.1109/iwcmc55113.2022.9824980}},
  year         = {{2022}},
}

@inproceedings{33807,
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}},
  publisher    = {{IEEE}},
  title        = {{{On Synchronization of Wireless Acoustic Sensor Networks in the Presence of Time-Varying Sampling Rate Offsets and Speaker Changes}}},
  doi          = {{10.1109/icassp43922.2022.9746284}},
  year         = {{2022}},
}

@inproceedings{33808,
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Heitkaemper, Jens and Haeb-Umbach, Reinhold}},
  booktitle    = {{2022 International Workshop on Acoustic Signal Enhancement (IWAENC)}},
  location     = {{ Bamberg, Germany }},
  publisher    = {{IEEE}},
  title        = {{{Informed vs. Blind Beamforming in Ad-Hoc Acoustic Sensor Networks for Meeting Transcription}}},
  doi          = {{10.1109/IWAENC53105.2022.9914772}},
  year         = {{2022}},
}

@misc{33816,
  author       = {{Gburrek, Tobias and Boeddeker, Christoph and von Neumann, Thilo and Cord-Landwehr, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  publisher    = {{arXiv}},
  title        = {{{A Meeting Transcription System for an Ad-Hoc Acoustic Sensor Network}}},
  doi          = {{10.48550/ARXIV.2205.00944}},
  year         = {{2022}},
}

@inproceedings{24000,
  author       = {{Heitkaemper, Jens and Schmalenstroeer, Joerg and Ion, Valentin and Haeb-Umbach, Reinhold}},
  booktitle    = {{Speech Communication; 14th ITG-Symposium}},
  pages        = {{1--5}},
  title        = {{{A Database for Research on Detection and Enhancement of Speech Transmitted over HF links}}},
  year         = {{2021}},
}

@inproceedings{23998,
  author       = {{Schmalenstroeer, Joerg and Heitkaemper, Jens and Ullmann, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{29th European Signal Processing Conference (EUSIPCO)}},
  pages        = {{1--5}},
  title        = {{{Open Range Pitch Tracking for Carrier Frequency Difference Estimation from HF Transmitted Speech}}},
  year         = {{2021}},
}

@article{22528,
  abstract     = {{Due to the ad hoc nature of wireless acoustic sensor networks, the position of the sensor nodes is typically unknown. This contribution proposes a technique to estimate the position and orientation of the sensor nodes from the recorded speech signals. The method assumes that a node comprises a microphone array with synchronously sampled microphones rather than a single microphone, but does not require the sampling clocks of the nodes to be synchronized. From the observed audio signals, the distances between the acoustic sources and arrays, as well as the directions of arrival, are estimated. They serve as input to a non-linear least squares problem, from which both the sensor nodes’ positions and orientations, as well as the source positions, are alternatingly estimated in an iterative process. Given one set of unknowns, i.e., either the source positions or the sensor nodes’ geometry, the other set of unknowns can be computed in closed-form. The proposed approach is computationally efficient and the first one, which employs both distance and directional information for geometry calibration in a common cost function. Since both distance and direction of arrival measurements suffer from outliers, e.g., caused by strong reflections of the sound waves on the surfaces of the room, we introduce measures to deemphasize or remove unreliable measurements. Additionally, we discuss modifications of our previously proposed deep neural network-based acoustic distance estimator, to account not only for omnidirectional sources but also for directional sources. Simulation results show good positioning accuracy and compare very favorably with alternative approaches from the literature.}},
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  issn         = {{1687-4722}},
  journal      = {{EURASIP Journal on Audio, Speech, and Music Processing}},
  title        = {{{Geometry calibration in wireless acoustic sensor networks utilizing DoA and distance information}}},
  doi          = {{10.1186/s13636-021-00210-x}},
  year         = {{2021}},
}

@inproceedings{23994,
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}},
  title        = {{{Iterative Geometry Calibration from Distance Estimates for Wireless Acoustic Sensor Networks}}},
  doi          = {{10.1109/icassp39728.2021.9413831}},
  year         = {{2021}},
}

@inproceedings{23999,
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{Speech Communication; 14th ITG-Symposium}},
  pages        = {{1--5}},
  title        = {{{On Source-Microphone Distance Estimation Using Convolutional Recurrent Neural Networks}}},
  year         = {{2021}},
}

@inproceedings{23997,
  author       = {{Chinaev, Aleksej and Enzner, Gerald and Gburrek, Tobias and Schmalenstroeer, Joerg}},
  booktitle    = {{29th European Signal Processing Conference (EUSIPCO)}},
  pages        = {{1--5}},
  title        = {{{Online Estimation of Sampling Rate Offsets in Wireless Acoustic Sensor Networks with Packet Loss}}},
  year         = {{2021}},
}

@inproceedings{20505,
  abstract     = {{Speech activity detection (SAD), which often rests on the fact that the noise is "more'' stationary than speech, is particularly challenging in non-stationary environments, because the time variance of the acoustic scene makes it difficult to discriminate  speech from noise. We propose two approaches to SAD, where one is based on statistical signal processing, while the other utilizes neural networks. The former employs sophisticated signal processing to track the noise and speech energies and is meant to support the case for a resource efficient, unsupervised signal processing approach.
The latter introduces a recurrent network layer that operates on short segments of the input speech to do temporal smoothing in the presence of non-stationary noise. The systems are tested on the Fearless Steps challenge database, which consists of the transmission data from the Apollo-11 space mission.
The statistical SAD  achieves comparable detection performance to earlier proposed neural network based SADs, while the neural network based approach leads to a decision cost function of 1.07% on the evaluation set of the 2020 Fearless Steps Challenge, which sets a new state of the art.}},
  author       = {{Heitkaemper, Jens and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{INTERSPEECH 2020 Virtual Shanghai China}},
  keywords     = {{voice activity detection, speech activity detection, neural network, statistical speech processing}},
  title        = {{{Statistical and Neural Network Based Speech Activity Detection in Non-Stationary Acoustic Environments}}},
  year         = {{2020}},
}

@inproceedings{18651,
  abstract     = {{We present an approach to deep neural network based (DNN-based) distance estimation in reverberant rooms for supporting geometry calibration tasks in wireless acoustic sensor networks. Signal diffuseness information from acoustic signals is aggregated via the coherent-to-diffuse power ratio to obtain a distance-related feature, which is mapped to a source-to-microphone distance estimate by means of a DNN. This information is then combined with direction-of-arrival estimates from compact microphone arrays to infer the geometry of the sensor network. Unlike many other approaches to geometry calibration, the proposed scheme does only require that the sampling clocks of the sensor nodes are roughly synchronized. In simulations we show that the proposed DNN-based distance estimator generalizes to unseen acoustic environments and that precise estimates of the sensor node positions are obtained. }},
  author       = {{Gburrek, Tobias and Schmalenstroeer, Joerg and Brendel, Andreas and Kellermann, Walter and Haeb-Umbach, Reinhold}},
  booktitle    = {{European Signal Processing Conference (EUSIPCO)}},
  title        = {{{Deep Neural Network based Distance Estimation for Geometry Calibration in Acoustic Sensor Network}}},
  year         = {{2020}},
}

@inproceedings{12899,
  abstract     = {{This contribution presents a speech enhancement system for the CHiME-5 Dinner Party Scenario. The front-end employs multi-channel linear time-variant filtering and achieves its gains without the use of a neural network. We present an adaptation of blind source separation techniques to the CHiME-5 database which we call Guided Source Separation (GSS). Using the baseline acoustic and language model, the combination of Weighted Prediction Error based dereverberation, guided source separation, and beamforming reduces the WER by 10:54% (relative) for the single array track and by 21:12% (relative) on the multiple array track.}},
  author       = {{Boeddeker, Christoph and Heitkaemper, Jens and Schmalenstroeer, Joerg and Drude, Lukas and Heymann, Jahn and Haeb-Umbach, Reinhold}},
  booktitle    = {{Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India}},
  title        = {{{Front-End Processing for the CHiME-5 Dinner Party Scenario}}},
  year         = {{2018}},
}

@inproceedings{6859,
  abstract     = {{Signal processing in WASNs is based on a software framework for hosting the algorithms as well as on a set of wireless connected devices representing the hardware. Each of the nodes contributes memory, processing power, communication bandwidth and some sensor information for the tasks to be solved on the network. 
In this paper we present our MARVELO framework for distributed signal processing. It is intended for transforming existing centralized implementations into distributed versions. To this end, the software only needs a block-oriented implementation, which MARVELO picks-up and distributes on the network. Additionally, our sensor node hardware and the audio interfaces responsible for multi-channel recordings are presented.}},
  author       = {{Afifi, Haitham and Schmalenstroeer, Joerg and Ullmann, Joerg and Haeb-Umbach, Reinhold and Karl, Holger}},
  booktitle    = {{Speech Communication; 13th ITG-Symposium}},
  pages        = {{1--5}},
  title        = {{{MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks}}},
  year         = {{2018}},
}

@inproceedings{11838,
  abstract     = {{Distributed sensor data acquisition usually encompasses data sampling by the individual devices, where each of them has its own oscillator driving the local sampling process, resulting in slightly different sampling rates at the individual sensor nodes. Nevertheless, for certain downstream signal processing tasks it is important to compensate even for small sampling rate offsets. Aligning the sampling rates of oscillators which differ only by a few parts-per-million, is, however, challenging and quite different from traditional multirate signal processing tasks. In this paper we propose to transfer a precise but computationally demanding time domain approach, inspired by the Nyquist-Shannon sampling theorem, to an efficient frequency domain implementation. To this end a buffer control is employed which compensates for sampling offsets which are multiples of the sampling period, while a digital filter, realized by the wellknown Overlap-Save method, handles the fractional part of the sampling phase offset. With experiments on artificially misaligned data we investigate the parametrization, the efficiency, and the induced distortions of the proposed resampling method. It is shown that a favorable compromise between residual distortion and computational complexity is achieved, compared to other sampling rate offset compensation techniques.}},
  author       = {{Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{26th European Signal Processing Conference (EUSIPCO 2018)}},
  title        = {{{Efficient Sampling Rate Offset Compensation - An Overlap-Save Based Approach}}},
  year         = {{2018}},
}

@inproceedings{11876,
  abstract     = {{This paper describes the systems for the single-array track and the multiple-array track of the 5th CHiME Challenge. The final system is a combination of multiple systems, using Confusion Network Combination (CNC). The different systems presented here are utilizing different front-ends and training sets for a Bidirectional Long Short-Term Memory (BLSTM) Acoustic Model (AM). The front-end was replaced by enhancements provided by Paderborn University [1]. The back-end has been implemented using RASR [2] and RETURNN [3]. Additionally, a system combination including the hypothesis word graphs from the system of the submission [1] has been performed, which results in the final best system.}},
  author       = {{Kitza, Markus and Michel, Wilfried and Boeddeker, Christoph and Heitkaemper, Jens and Menne, Tobias and Schlüter, Ralf and Ney, Hermann and Schmalenstroeer, Joerg and Drude, Lukas and Heymann, Jahn and Haeb-Umbach, Reinhold}},
  booktitle    = {{Proc. CHiME 2018 Workshop on Speech Processing in Everyday Environments, Hyderabad, India}},
  title        = {{{The RWTH/UPB System Combination for the CHiME 2018 Workshop}}},
  year         = {{2018}},
}

@inproceedings{11836,
  abstract     = {{Due to their distributed nature wireless acoustic sensor networks offer great potential for improved signal acquisition, processing and classification for applications such as monitoring and surveillance, home automation, or hands-free telecommunication. To reduce the communication demand with a central server and to raise the privacy level it is desirable to perform processing at node level. The limited processing and memory capabilities on a sensor node, however, stand in contrast to the compute and memory intensive deep learning algorithms used in modern speech and audio processing. In this work, we perform benchmarking of commonly used convolutional and recurrent neural network architectures on a Raspberry Pi based acoustic sensor node. We show that it is possible to run medium-sized neural network topologies used for speech enhancement and speech recognition in real time. For acoustic event recognition, where predictions in a lower temporal resolution are sufficient, it is even possible to run current state-of-the-art deep convolutional models with a real-time-factor of 0:11.}},
  author       = {{Ebbers, Janek and Heitkaemper, Jens and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{ITG 2018, Oldenburg, Germany}},
  title        = {{{Benchmarking Neural Network Architectures for Acoustic Sensor Networks}}},
  year         = {{2018}},
}

@inproceedings{11839,
  abstract     = {{It has been experimentally verified that sampling rate offsets (SROs) between the input channels of an acoustic beamformer have a detrimental effect on the achievable SNR gains. In this paper we derive an analytic model to study the impact of SRO on the estimation of the spatial noise covariance matrix used in MVDR beamforming. It is shown that a perfect compensation of the SRO is impossible if the noise covariance matrix is estimated by time averaging, even if the SRO is perfectly known. The SRO should therefore be compensated for prior to beamformer coefficient estimation. We present a novel scheme where SRO compensation and beamforming closely interact, saving some computational effort compared to separate SRO adjustment followed by acoustic beamforming.}},
  author       = {{Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}},
  booktitle    = {{ITG 2018, Oldenburg, Germany}},
  title        = {{{Insights into the Interplay of Sampling Rate Offsets and MVDR Beamforming}}},
  year         = {{2018}},
}

@inproceedings{15952,
  abstract     = {{Arbitrary sampling rate conversion has already received considerable attention in the past, but still lacks an equivalent representation of the effective time-dilation process in the block frequency domain. Good sampling rate converters in the time domain have been known, for instance, in terms of time-varying 'Sinc' or fixed 'Farrow' polynomial filters. The former can deliver nearly exact conversion at high complexity, while the latter has pronounced computational efficiency with limited accuracy. Only recently, it was shown that a composite 'polyphase Farrow' form with high resampling precision can be implemented with quasi-fixed filters that operate at the input sampling rate. We therefore propose to capitalize from that fixed-filter architecture in that we translate the polyphase-Farrow filters into an equivalent FFT-based overlap-save form. Experimental evaluation and comparison with other state-of-the art frequency-domain approaches then proves currently the best price-performance ratio of the proposed algorithm. It is thus an ideal candidate for the new framework of acoustic sensor networks that critically rests upon fast and accurate alignment of autonomous sampling processes.}},
  author       = {{Schmalenstroeer, Joerg and Chinaev, Aleksej and Enzner, Gerald}},
  booktitle    = {{Speech Communication; 13th ITG-Symposium}},
  issn         = {{null}},
  pages        = {{1--5}},
  title        = {{{Fast and Accurate Audio Resampling for Acoustic Sensor Networks by Polyphase-Farrow Filters with FFT Realization}}},
  year         = {{2018}},
}

@misc{12081,
  abstract     = {{The invention relates to a building or enclosure termination opening and/or closing apparatus having communication signed or encrypted by means of a key, and to a method for operating such. To allow simple, convenient and secure use by exclusively authorised users, the apparatus comprises: a first and a second user terminal, with secure forwarding of a time-limited key from the first to the second user terminal being possible. According to an alternative, individual keys are generated by a user identification and a secret device key.}},
  author       = {{Jacob, Florian and Schmalenstroeer, Joerg}},
  title        = {{{Building or Enclosure Termination Closing and/or Opening Apparatus, and Method for Operating a Building or Enclosure Termination}}},
  year         = {{2017}},
}

