[{"publication":"ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","abstract":[{"text":"Sound capture by microphone arrays opens the possibility to exploit spatial, in addition to spectral, information for diarization and signal enhancement, two important tasks in meeting transcription. However, there is no one-to-one mapping of positions in space to speakers if speakers move. Here, we address this by proposing a novel joint spatial and spectral mixture model, whose two submodels are loosely coupled by modeling the relationship between speaker and position index probabilistically. Thus, spatial and spectral information can be jointly exploited, while at the same time allowing for speakers speaking from different positions. Experiments on the LibriCSS data set with simulated speaker position changes show great improvements over tightly coupled subsystems.","lang":"eng"}],"date_created":"2026-05-11T14:20:48Z","department":[{"_id":"54"},{"_id":"1063"}],"type":"conference","keyword":["mixture models","meeting processing","diarization","source separation"],"author":[{"full_name":"Meise, Adrian Tobias","first_name":"Adrian Tobias","last_name":"Meise","id":"79268"},{"last_name":"Cord-Landwehr","first_name":"Tobias","full_name":"Cord-Landwehr, Tobias","id":"44393"},{"first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph","id":"40767"},{"last_name":"Delcroix","first_name":"Marc","full_name":"Delcroix, Marc"},{"last_name":"Nakatani","first_name":"Tomohiro","full_name":"Nakatani, Tomohiro"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"}],"title":"Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments","year":"2026","date_updated":"2026-05-15T08:17:25Z","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://arxiv.org/pdf/2601.16077","open_access":"1"}],"doi":"10.1109/icassp55912.2026.11463540","citation":{"short":"A.T. Meise, T. Cord-Landwehr, C. Boeddeker, M. Delcroix, T. Nakatani, R. Haeb-Umbach, in: ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2026.","chicago":"Meise, Adrian Tobias, Tobias Cord-Landwehr, Christoph Boeddeker, Marc Delcroix, Tomohiro Nakatani, and Reinhold Haeb-Umbach. “Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments.” In <i>ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE, 2026. <a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">https://doi.org/10.1109/icassp55912.2026.11463540</a>.","ieee":"A. T. Meise, T. Cord-Landwehr, C. Boeddeker, M. Delcroix, T. Nakatani, and R. Haeb-Umbach, “Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments,” presented at the  2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) , Barcelona, 2026, doi: <a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">10.1109/icassp55912.2026.11463540</a>.","apa":"Meise, A. T., Cord-Landwehr, T., Boeddeker, C., Delcroix, M., Nakatani, T., &#38; Haeb-Umbach, R. (2026). Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments. <i>ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>.  2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) , Barcelona. <a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">https://doi.org/10.1109/icassp55912.2026.11463540</a>","bibtex":"@inproceedings{Meise_Cord-Landwehr_Boeddeker_Delcroix_Nakatani_Haeb-Umbach_2026, title={Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments}, DOI={<a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">10.1109/icassp55912.2026.11463540</a>}, booktitle={ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, publisher={IEEE}, author={Meise, Adrian Tobias and Cord-Landwehr, Tobias and Boeddeker, Christoph and Delcroix, Marc and Nakatani, Tomohiro and Haeb-Umbach, Reinhold}, year={2026} }","ama":"Meise AT, Cord-Landwehr T, Boeddeker C, Delcroix M, Nakatani T, Haeb-Umbach R. Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments. In: <i>ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE; 2026. doi:<a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">10.1109/icassp55912.2026.11463540</a>","mla":"Meise, Adrian Tobias, et al. “Loose Coupling of Spectral and Spatial Models for Multi-Channel Diarization and Enhancement of Meetings in Dynamic Environments.” <i>ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, IEEE, 2026, doi:<a href=\"https://doi.org/10.1109/icassp55912.2026.11463540\">10.1109/icassp55912.2026.11463540</a>."},"project":[{"_id":"52","name":"Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"external_id":{"arxiv":["https://arxiv.org/abs/2601.16077"]},"oa":"1","conference":{"location":"Barcelona","name":" 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) "},"status":"public","publisher":"IEEE","_id":"65606","user_id":"79268"},{"department":[{"_id":"54"}],"type":"conference","keyword":["Bayes methods","Gaussian processes","convolution","decision theory","decoding","noise","reverberation","speech coding","speech recognition","Bayesian decision rule","GMM","Gaussian mixture models","additive noise scenarios","automatic speech recognition systems","convolutive noise scenarios","decoding approach","mathematical framework","reverberant environments","significance decoding","speech feature estimation","uncertainty-of-observation techniques","Hidden Markov models","Maximum likelihood decoding","Noise","Speech","Speech recognition","Uncertainty","Uncertainty-of-observation","modified imputation","noise robust speech recognition","significance decoding","uncertainty decoding"],"date_created":"2019-07-12T05:26:53Z","abstract":[{"lang":"eng","text":"The accuracy of automatic speech recognition systems in noisy and reverberant environments can be improved notably by exploiting the uncertainty of the estimated speech features using so-called uncertainty-of-observation techniques. In this paper, we introduce a new Bayesian decision rule that can serve as a mathematical framework from which both known and new uncertainty-of-observation techniques can be either derived or approximated. The new decision rule in its direct form leads to the new significance decoding approach for Gaussian mixture models, which results in better performance compared to standard uncertainty-of-observation techniques in different additive and convolutive noise scenarios."}],"citation":{"bibtex":"@inproceedings{Abdelaziz_Zeiler_Kolossa_Leutnant_Haeb-Umbach_2013, title={GMM-based significance decoding}, DOI={<a href=\"https://doi.org/10.1109/ICASSP.2013.6638984\">10.1109/ICASSP.2013.6638984</a>}, booktitle={Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on}, author={Abdelaziz, Ahmed H. and Zeiler, Steffen and Kolossa, Dorothea and Leutnant, Volker and Haeb-Umbach, Reinhold}, year={2013}, pages={6827–6831} }","ama":"Abdelaziz AH, Zeiler S, Kolossa D, Leutnant V, Haeb-Umbach R. GMM-based significance decoding. In: <i>Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference On</i>. ; 2013:6827-6831. doi:<a href=\"https://doi.org/10.1109/ICASSP.2013.6638984\">10.1109/ICASSP.2013.6638984</a>","mla":"Abdelaziz, Ahmed H., et al. “GMM-Based Significance Decoding.” <i>Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference On</i>, 2013, pp. 6827–31, doi:<a href=\"https://doi.org/10.1109/ICASSP.2013.6638984\">10.1109/ICASSP.2013.6638984</a>.","chicago":"Abdelaziz, Ahmed H., Steffen Zeiler, Dorothea Kolossa, Volker Leutnant, and Reinhold Haeb-Umbach. “GMM-Based Significance Decoding.” In <i>Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference On</i>, 6827–31, 2013. <a href=\"https://doi.org/10.1109/ICASSP.2013.6638984\">https://doi.org/10.1109/ICASSP.2013.6638984</a>.","short":"A.H. Abdelaziz, S. Zeiler, D. Kolossa, V. Leutnant, R. Haeb-Umbach, in: Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference On, 2013, pp. 6827–6831.","ieee":"A. H. Abdelaziz, S. Zeiler, D. Kolossa, V. Leutnant, and R. Haeb-Umbach, “GMM-based significance decoding,” in <i>Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on</i>, 2013, pp. 6827–6831.","apa":"Abdelaziz, A. H., Zeiler, S., Kolossa, D., Leutnant, V., &#38; Haeb-Umbach, R. (2013). GMM-based significance decoding. In <i>Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on</i> (pp. 6827–6831). <a href=\"https://doi.org/10.1109/ICASSP.2013.6638984\">https://doi.org/10.1109/ICASSP.2013.6638984</a>"},"publication":"Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on","doi":"10.1109/ICASSP.2013.6638984","user_id":"44006","language":[{"iso":"eng"}],"_id":"11716","page":"6827-6831","date_updated":"2022-01-06T06:51:07Z","publication_identifier":{"issn":["1520-6149"]},"author":[{"first_name":"Ahmed H.","last_name":"Abdelaziz","full_name":"Abdelaziz, Ahmed H."},{"last_name":"Zeiler","first_name":"Steffen","full_name":"Zeiler, Steffen"},{"full_name":"Kolossa, Dorothea","last_name":"Kolossa","first_name":"Dorothea"},{"last_name":"Leutnant","first_name":"Volker","full_name":"Leutnant, Volker"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}],"year":"2013","title":"GMM-based significance decoding","status":"public"}]
