[{"quality_controlled":"1","publication":"ITG Conference on Speech Communication","file_date_updated":"2023-11-15T14:48:44Z","citation":{"apa":"Schmalenstroeer, J., Gburrek, T., &#38; Haeb-Umbach, R. (2023). LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices. <i>ITG Conference on Speech Communication</i>. ITG Conference on Speech Communication, Aachen.","ieee":"J. Schmalenstroeer, T. Gburrek, and R. Haeb-Umbach, “LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices,” presented at the ITG Conference on Speech Communication, Aachen, 2023.","chicago":"Schmalenstroeer, Joerg, Tobias Gburrek, and Reinhold Haeb-Umbach. “LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices.” In <i>ITG Conference on Speech Communication</i>, 2023.","short":"J. Schmalenstroeer, T. Gburrek, R. Haeb-Umbach, in: ITG Conference on Speech Communication, 2023.","mla":"Schmalenstroeer, Joerg, et al. “LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices.” <i>ITG Conference on Speech Communication</i>, 2023.","ama":"Schmalenstroeer J, Gburrek T, Haeb-Umbach R. LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices. In: <i>ITG Conference on Speech Communication</i>. ; 2023.","bibtex":"@inproceedings{Schmalenstroeer_Gburrek_Haeb-Umbach_2023, title={LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices}, booktitle={ITG Conference on Speech Communication}, author={Schmalenstroeer, Joerg and Gburrek, Tobias and Haeb-Umbach, Reinhold}, year={2023} }"},"type":"conference","oa":"1","department":[{"_id":"54"}],"file":[{"file_id":"48483","content_type":"application/pdf","file_name":"SchTgbHaeb2023Final.pdf","access_level":"open_access","file_size":2844502,"relation":"main_file","date_updated":"2023-11-15T14:48:44Z","date_created":"2023-10-26T08:20:15Z","creator":"schmalen"}],"date_created":"2023-10-18T13:00:54Z","date_updated":"2023-11-15T14:48:45Z","has_accepted_license":"1","year":"2023","status":"public","title":"LibriWASN: A Data Set for Meeting Separation, Diarization, and Recognition with Asynchronous Recording Devices","conference":{"location":"Aachen","name":"ITG Conference on Speech Communication"},"author":[{"full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer","id":"460"},{"last_name":"Gburrek","first_name":"Tobias","full_name":"Gburrek, Tobias","id":"44006"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"ddc":["004"],"user_id":"460","language":[{"iso":"eng"}],"_id":"48270"},{"main_file_link":[{"open_access":"1","url":"https://arxiv.org/abs/2310.12599"}],"language":[{"iso":"eng"}],"date_updated":"2023-11-22T13:44:33Z","title":"On Feature Importance and Interpretability of Speaker Representations","year":"2023","author":[{"id":"72602","full_name":"Rautenberg, Frederik","first_name":"Frederik","last_name":"Rautenberg"},{"id":"49871","full_name":"Kuhlmann, Michael","first_name":"Michael","last_name":"Kuhlmann"},{"first_name":"Jana","last_name":"Wiechmann","full_name":"Wiechmann, Jana"},{"last_name":"Seebauer","first_name":"Fritz","full_name":"Seebauer, Fritz"},{"first_name":"Petra","last_name":"Wagner","full_name":"Wagner, Petra"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"type":"conference","department":[{"_id":"54"},{"_id":"660"}],"file":[{"relation":"main_file","date_updated":"2023-10-20T08:20:58Z","file_name":"arxiv.pdf","file_size":272390,"access_level":"closed","file_id":"48359","success":1,"content_type":"application/pdf","creator":"frra","date_created":"2023-10-20T08:20:58Z"}],"date_created":"2023-10-20T08:04:46Z","abstract":[{"text":"Unsupervised speech disentanglement aims at separating fast varying from\r\nslowly varying components of a speech signal. In this contribution, we take a\r\ncloser look at the embedding vector representing the slowly varying signal\r\ncomponents, commonly named the speaker embedding vector. We ask, which\r\nproperties of a speaker's voice are captured and investigate to which extent do\r\nindividual embedding vector components sign responsible for them, using the\r\nconcept of Shapley values. Our findings show that certain speaker-specific\r\nacoustic-phonetic properties can be fairly well predicted from the speaker\r\nembedding, while the investigated more abstract voice quality features cannot.","lang":"eng"}],"publication":"ITG Conference on Speech Communication","ddc":["000"],"user_id":"72602","_id":"48355","has_accepted_license":"1","status":"public","conference":{"start_date":"2023-09-20","name":"ITG Conference on Speech Communication","location":"Aachen","end_date":"2023-09-22"},"oa":"1","external_id":{"arxiv":["2310.12599"]},"project":[{"grant_number":"438445824","_id":"129","name":"TRR 318 - C06: TRR 318 - Technisch unterstütztes Erklären von Stimmcharakteristika (Teilprojekt C06)"}],"file_date_updated":"2023-10-20T08:20:58Z","citation":{"mla":"Rautenberg, Frederik, et al. “On Feature Importance and Interpretability of Speaker Representations.” <i>ITG Conference on Speech Communication</i>, 2023.","bibtex":"@inproceedings{Rautenberg_Kuhlmann_Wiechmann_Seebauer_Wagner_Haeb-Umbach_2023, title={On Feature Importance and Interpretability of Speaker Representations}, booktitle={ITG Conference on Speech Communication}, author={Rautenberg, Frederik and Kuhlmann, Michael and Wiechmann, Jana and Seebauer, Fritz and Wagner, Petra and Haeb-Umbach, Reinhold}, year={2023} }","ama":"Rautenberg F, Kuhlmann M, Wiechmann J, Seebauer F, Wagner P, Haeb-Umbach R. On Feature Importance and Interpretability of Speaker Representations. In: <i>ITG Conference on Speech Communication</i>. ; 2023.","ieee":"F. Rautenberg, M. Kuhlmann, J. Wiechmann, F. Seebauer, P. Wagner, and R. Haeb-Umbach, “On Feature Importance and Interpretability of Speaker Representations,” presented at the ITG Conference on Speech Communication, Aachen, 2023.","apa":"Rautenberg, F., Kuhlmann, M., Wiechmann, J., Seebauer, F., Wagner, P., &#38; Haeb-Umbach, R. (2023). On Feature Importance and Interpretability of Speaker Representations. <i>ITG Conference on Speech Communication</i>. ITG Conference on Speech Communication, Aachen.","short":"F. Rautenberg, M. Kuhlmann, J. Wiechmann, F. Seebauer, P. Wagner, R. Haeb-Umbach, in: ITG Conference on Speech Communication, 2023.","chicago":"Rautenberg, Frederik, Michael Kuhlmann, Jana Wiechmann, Fritz Seebauer, Petra Wagner, and Reinhold Haeb-Umbach. “On Feature Importance and Interpretability of Speaker Representations.” In <i>ITG Conference on Speech Communication</i>, 2023."}},{"publication":"20th International Congress of the Phonetic Sciences (ICPhS) ","type":"conference","department":[{"_id":"54"},{"_id":"660"}],"file":[{"relation":"main_file","date_updated":"2023-10-24T08:03:27Z","file_name":"188.pdf","file_size":209980,"access_level":"closed","file_id":"48413","success":1,"content_type":"application/pdf","creator":"frra","date_created":"2023-10-24T08:03:27Z"}],"date_created":"2023-10-24T08:05:40Z","date_updated":"2023-11-22T13:44:59Z","year":"2023","title":"Explaining voice characteristics to novice voice practitioners-How successful is it?","author":[{"full_name":"Wiechmann, Jana","last_name":"Wiechmann","first_name":"Jana"},{"id":"72602","full_name":"Rautenberg, Frederik","first_name":"Frederik","last_name":"Rautenberg"},{"first_name":"Petra","last_name":"Wagner","full_name":"Wagner, Petra"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"main_file_link":[{"open_access":"1"}],"language":[{"iso":"eng"}],"project":[{"grant_number":"438445824","_id":"129","name":"TRR 318 - C06: TRR 318 - Technisch unterstütztes Erklären von Stimmcharakteristika (Teilprojekt C06)"}],"file_date_updated":"2023-10-24T08:03:27Z","citation":{"short":"J. Wiechmann, F. Rautenberg, P. Wagner, R. Haeb-Umbach, in: 20th International Congress of the Phonetic Sciences (ICPhS) , 2023.","chicago":"Wiechmann, Jana, Frederik Rautenberg, Petra Wagner, and Reinhold Haeb-Umbach. “Explaining Voice Characteristics to Novice Voice Practitioners-How Successful Is It?” In <i>20th International Congress of the Phonetic Sciences (ICPhS) </i>, 2023.","ieee":"J. Wiechmann, F. Rautenberg, P. Wagner, and R. Haeb-Umbach, “Explaining voice characteristics to novice voice practitioners-How successful is it?,” 2023.","apa":"Wiechmann, J., Rautenberg, F., Wagner, P., &#38; Haeb-Umbach, R. (2023). Explaining voice characteristics to novice voice practitioners-How successful is it? <i>20th International Congress of the Phonetic Sciences (ICPhS) </i>.","bibtex":"@inproceedings{Wiechmann_Rautenberg_Wagner_Haeb-Umbach_2023, title={Explaining voice characteristics to novice voice practitioners-How successful is it?}, booktitle={20th International Congress of the Phonetic Sciences (ICPhS) }, author={Wiechmann, Jana and Rautenberg, Frederik and Wagner, Petra and Haeb-Umbach, Reinhold}, year={2023} }","ama":"Wiechmann J, Rautenberg F, Wagner P, Haeb-Umbach R. Explaining voice characteristics to novice voice practitioners-How successful is it? In: <i>20th International Congress of the Phonetic Sciences (ICPhS) </i>. ; 2023.","mla":"Wiechmann, Jana, et al. “Explaining Voice Characteristics to Novice Voice Practitioners-How Successful Is It?” <i>20th International Congress of the Phonetic Sciences (ICPhS) </i>, 2023."},"oa":"1","has_accepted_license":"1","status":"public","conference":{"start_date":"2023-08-07","end_date":"2023-08-11"},"ddc":["040"],"user_id":"72602","_id":"48410"},{"publication":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","citation":{"chicago":"Aralikatti, Rohith, Christoph Boeddeker, Gordon Wichern, Aswin Subramanian, and Jonathan Le Roux. “Reverberation as Supervision For Speech Separation.” In <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE, 2023. <a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">https://doi.org/10.1109/icassp49357.2023.10095022</a>.","short":"R. Aralikatti, C. Boeddeker, G. Wichern, A. Subramanian, J. Le Roux, in: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2023.","apa":"Aralikatti, R., Boeddeker, C., Wichern, G., Subramanian, A., &#38; Le Roux, J. (2023). Reverberation as Supervision For Speech Separation. <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. <a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">https://doi.org/10.1109/icassp49357.2023.10095022</a>","ieee":"R. Aralikatti, C. Boeddeker, G. Wichern, A. Subramanian, and J. Le Roux, “Reverberation as Supervision For Speech Separation,” 2023, doi: <a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">10.1109/icassp49357.2023.10095022</a>.","ama":"Aralikatti R, Boeddeker C, Wichern G, Subramanian A, Le Roux J. Reverberation as Supervision For Speech Separation. In: <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE; 2023. doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">10.1109/icassp49357.2023.10095022</a>","bibtex":"@inproceedings{Aralikatti_Boeddeker_Wichern_Subramanian_Le Roux_2023, title={Reverberation as Supervision For Speech Separation}, DOI={<a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">10.1109/icassp49357.2023.10095022</a>}, booktitle={ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, publisher={IEEE}, author={Aralikatti, Rohith and Boeddeker, Christoph and Wichern, Gordon and Subramanian, Aswin and Le Roux, Jonathan}, year={2023} }","mla":"Aralikatti, Rohith, et al. “Reverberation as Supervision For Speech Separation.” <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, IEEE, 2023, doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10095022\">10.1109/icassp49357.2023.10095022</a>."},"type":"conference","department":[{"_id":"54"}],"date_created":"2023-10-23T15:09:13Z","publication_status":"published","date_updated":"2023-10-23T15:10:16Z","title":"Reverberation as Supervision For Speech Separation","year":"2023","status":"public","author":[{"full_name":"Aralikatti, Rohith","last_name":"Aralikatti","first_name":"Rohith"},{"id":"40767","last_name":"Boeddeker","first_name":"Christoph","full_name":"Boeddeker, Christoph"},{"full_name":"Wichern, Gordon","first_name":"Gordon","last_name":"Wichern"},{"last_name":"Subramanian","first_name":"Aswin","full_name":"Subramanian, Aswin"},{"full_name":"Le Roux, Jonathan","first_name":"Jonathan","last_name":"Le Roux"}],"user_id":"40767","doi":"10.1109/icassp49357.2023.10095022","_id":"48391","language":[{"iso":"eng"}],"publisher":"IEEE"},{"department":[{"_id":"54"}],"type":"conference","date_created":"2023-07-15T16:10:20Z","project":[{"name":"TRR 318 - C06: TRR 318 - Technisch unterstütztes Erklären von Stimmcharakteristika (Teilprojekt C06)","_id":"129","grant_number":"438445824"}],"citation":{"mla":"Seebauer, Fritz, et al. “Re-Examining the Quality Dimensions of Synthetic Speech.” <i>12th Speech Synthesis Workshop (SSW) 2023</i>, 2023.","ama":"Seebauer F, Kuhlmann M, Haeb-Umbach R, Wagner P. Re-examining the quality dimensions of synthetic speech. In: <i>12th Speech Synthesis Workshop (SSW) 2023</i>. ; 2023.","bibtex":"@inproceedings{Seebauer_Kuhlmann_Haeb-Umbach_Wagner_2023, title={Re-examining the quality dimensions of synthetic speech}, booktitle={12th Speech Synthesis Workshop (SSW) 2023}, author={Seebauer, Fritz and Kuhlmann, Michael and Haeb-Umbach, Reinhold and Wagner, Petra}, year={2023} }","apa":"Seebauer, F., Kuhlmann, M., Haeb-Umbach, R., &#38; Wagner, P. (2023). Re-examining the quality dimensions of synthetic speech. <i>12th Speech Synthesis Workshop (SSW) 2023</i>.","ieee":"F. Seebauer, M. Kuhlmann, R. Haeb-Umbach, and P. Wagner, “Re-examining the quality dimensions of synthetic speech,” 2023.","chicago":"Seebauer, Fritz, Michael Kuhlmann, Reinhold Haeb-Umbach, and Petra Wagner. “Re-Examining the Quality Dimensions of Synthetic Speech.” In <i>12th Speech Synthesis Workshop (SSW) 2023</i>, 2023.","short":"F. Seebauer, M. Kuhlmann, R. Haeb-Umbach, P. Wagner, in: 12th Speech Synthesis Workshop (SSW) 2023, 2023."},"publication":"12th Speech Synthesis Workshop (SSW) 2023","user_id":"242","language":[{"iso":"eng"}],"_id":"46069","has_accepted_license":"1","date_updated":"2023-10-25T08:42:56Z","author":[{"full_name":"Seebauer, Fritz","last_name":"Seebauer","first_name":"Fritz"},{"first_name":"Michael","last_name":"Kuhlmann","full_name":"Kuhlmann, Michael","id":"49871"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"},{"full_name":"Wagner, Petra","first_name":"Petra","last_name":"Wagner"}],"year":"2023","title":"Re-examining the quality dimensions of synthetic speech","status":"public"},{"publication":"IEEE/ACM Transactions on Audio, Speech, and Language Processing","abstract":[{"lang":"eng","text":"Continuous Speech Separation (CSS) has been proposed to address speech overlaps during the analysis of realistic meeting-like conversations by eliminating any overlaps before further processing.\r\nCSS separates a recording of arbitrarily many speakers into a small number of overlap-free output channels, where each output channel may contain speech of multiple speakers.\r\nThis is often done by applying a conventional separation model trained with Utterance-level Permutation Invariant Training (uPIT), which exclusively maps a speaker to an output channel, in sliding window approach called stitching.\r\nRecently, we introduced an alternative training scheme called Graph-PIT that teaches the separation network to directly produce output streams in the required format without stitching.\r\nIt can handle an arbitrary number of speakers as long as never more of them overlap at the same time than the separator has output channels.\r\nIn this contribution, we further investigate the Graph-PIT training scheme.\r\nWe show in extended experiments that models trained with Graph-PIT also work in challenging reverberant conditions.\r\nModels trained in this way are able to perform segment-less CSS, i.e., without stitching, and achieve comparable and often better separation quality than the conventional CSS with uPIT and stitching.\r\nWe simplify the training schedule for Graph-PIT with the recently proposed Source Aggregated Signal-to-Distortion Ratio (SA-SDR) loss.\r\nIt eliminates unfavorable properties of the previously used A-SDR loss and thus enables training with Graph-PIT from scratch.\r\nGraph-PIT training relaxes the constraints w.r.t. the allowed numbers of speakers and speaking patterns which allows using a larger variety of training data.\r\nFurthermore, we introduce novel signal-level evaluation metrics for meeting scenarios, namely the source-aggregated scale- and convolution-invariant Signal-to-Distortion Ratio (SA-SI-SDR and SA-CI-SDR), which are generalizations of the commonly used SDR-based metrics for the CSS case."}],"date_created":"2023-01-09T17:24:17Z","file":[{"date_created":"2023-01-09T17:46:05Z","creator":"haebumb","content_type":"application/pdf","file_id":"35607","date_updated":"2023-01-11T08:50:19Z","relation":"main_file","file_size":7185077,"access_level":"open_access","file_name":"main.pdf"}],"department":[{"_id":"54"}],"type":"journal_article","keyword":["Continuous Speech Separation","Source Separation","Graph-PIT","Dynamic Programming","Permutation Invariant Training"],"author":[{"full_name":"von Neumann, Thilo","first_name":"Thilo","orcid":"https://orcid.org/0000-0002-7717-8670","last_name":"von Neumann","id":"49870"},{"full_name":"Kinoshita, Keisuke","first_name":"Keisuke","last_name":"Kinoshita"},{"id":"40767","first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph"},{"first_name":"Marc","last_name":"Delcroix","full_name":"Delcroix, Marc"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"publication_identifier":{"issn":["2329-9290","2329-9304"]},"title":"Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria","year":"2023","article_type":"original","intvolume":"        31","publication_status":"published","date_updated":"2023-11-15T12:16:11Z","language":[{"iso":"eng"}],"doi":"10.1109/taslp.2022.3228629","citation":{"apa":"von Neumann, T., Kinoshita, K., Boeddeker, C., Delcroix, M., &#38; Haeb-Umbach, R. (2023). Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, <i>31</i>, 576–589. <a href=\"https://doi.org/10.1109/taslp.2022.3228629\">https://doi.org/10.1109/taslp.2022.3228629</a>","ieee":"T. von Neumann, K. Kinoshita, C. Boeddeker, M. Delcroix, and R. Haeb-Umbach, “Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria,” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, vol. 31, pp. 576–589, 2023, doi: <a href=\"https://doi.org/10.1109/taslp.2022.3228629\">10.1109/taslp.2022.3228629</a>.","short":"T. von Neumann, K. Kinoshita, C. Boeddeker, M. Delcroix, R. Haeb-Umbach, IEEE/ACM Transactions on Audio, Speech, and Language Processing 31 (2023) 576–589.","chicago":"Neumann, Thilo von, Keisuke Kinoshita, Christoph Boeddeker, Marc Delcroix, and Reinhold Haeb-Umbach. “Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i> 31 (2023): 576–89. <a href=\"https://doi.org/10.1109/taslp.2022.3228629\">https://doi.org/10.1109/taslp.2022.3228629</a>.","mla":"von Neumann, Thilo, et al. “Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, vol. 31, Institute of Electrical and Electronics Engineers (IEEE), 2023, pp. 576–89, doi:<a href=\"https://doi.org/10.1109/taslp.2022.3228629\">10.1109/taslp.2022.3228629</a>.","ama":"von Neumann T, Kinoshita K, Boeddeker C, Delcroix M, Haeb-Umbach R. Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>. 2023;31:576-589. doi:<a href=\"https://doi.org/10.1109/taslp.2022.3228629\">10.1109/taslp.2022.3228629</a>","bibtex":"@article{von Neumann_Kinoshita_Boeddeker_Delcroix_Haeb-Umbach_2023, title={Segment-Less Continuous Speech Separation of Meetings: Training and Evaluation Criteria}, volume={31}, DOI={<a href=\"https://doi.org/10.1109/taslp.2022.3228629\">10.1109/taslp.2022.3228629</a>}, journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing}, publisher={Institute of Electrical and Electronics Engineers (IEEE)}, author={von Neumann, Thilo and Kinoshita, Keisuke and Boeddeker, Christoph and Delcroix, Marc and Haeb-Umbach, Reinhold}, year={2023}, pages={576–589} }"},"file_date_updated":"2023-01-11T08:50:19Z","project":[{"_id":"52","name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"quality_controlled":"1","oa":"1","status":"public","has_accepted_license":"1","_id":"35602","publisher":"Institute of Electrical and Electronics Engineers (IEEE)","page":"576-589","volume":31,"user_id":"49870","ddc":["000"]},{"_id":"49109","ddc":["004"],"user_id":"460","status":"public","conference":{"start_date":"2023-10-31","name":"57th Asilomar Conference on Signals, Systems, and Computers","end_date":"2023-11-01"},"has_accepted_license":"1","oa":"1","file_date_updated":"2023-11-22T07:58:49Z","citation":{"ama":"Gburrek T, Schmalenstroeer J, Haeb-Umbach R. Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks. In: <i>Proc. Asilomar Conference on Signals, Systems, and Computers</i>. ; 2023.","bibtex":"@inproceedings{Gburrek_Schmalenstroeer_Haeb-Umbach_2023, title={Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks}, booktitle={Proc. Asilomar Conference on Signals, Systems, and Computers}, author={Gburrek, Tobias and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold}, year={2023} }","mla":"Gburrek, Tobias, et al. “Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks.” <i>Proc. Asilomar Conference on Signals, Systems, and Computers</i>, 2023.","chicago":"Gburrek, Tobias, Joerg Schmalenstroeer, and Reinhold Haeb-Umbach. “Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks.” In <i>Proc. Asilomar Conference on Signals, Systems, and Computers</i>, 2023.","short":"T. Gburrek, J. Schmalenstroeer, R. Haeb-Umbach, in: Proc. Asilomar Conference on Signals, Systems, and Computers, 2023.","apa":"Gburrek, T., Schmalenstroeer, J., &#38; Haeb-Umbach, R. (2023). Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks. <i>Proc. Asilomar Conference on Signals, Systems, and Computers</i>. 57th Asilomar Conference on Signals, Systems, and Computers.","ieee":"T. Gburrek, J. Schmalenstroeer, and R. Haeb-Umbach, “Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks,” presented at the 57th Asilomar Conference on Signals, Systems, and Computers, 2023."},"quality_controlled":"1","language":[{"iso":"eng"}],"title":"Spatial Diarization for Meeting Transcription with Ad-Hoc Acoustic Sensor Networks","year":"2023","author":[{"id":"44006","full_name":"Gburrek, Tobias","last_name":"Gburrek","first_name":"Tobias"},{"full_name":"Schmalenstroeer, Joerg","first_name":"Joerg","last_name":"Schmalenstroeer","id":"460"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"}],"date_updated":"2023-11-22T07:58:49Z","file":[{"date_created":"2023-11-22T07:51:18Z","creator":"schmalen","content_type":"application/pdf","file_id":"49110","date_updated":"2023-11-22T07:58:49Z","relation":"main_file","file_size":212317,"access_level":"open_access","file_name":"asilomar.pdf"}],"date_created":"2023-11-22T07:52:29Z","keyword":["Diarization","time difference of arrival","ad-hoc acoustic sensor network","meeting transcription"],"type":"conference","department":[{"_id":"54"}],"publication":"Proc. Asilomar Conference on Signals, Systems, and Computers","abstract":[{"lang":"eng","text":"We propose a diarization system, that estimates “who spoke when” based on spatial information, to be used as a front-end of a meeting transcription system running on the signals gathered from an acoustic sensor network (ASN). Although the\r\nspatial distribution of the microphones is advantageous, exploiting the spatial diversity for diarization and signal enhancement is challenging, because the microphones’ positions are typically unknown, and the recorded signals are initially unsynchronized in general. Here, we approach these issues by first blindly synchronizing the signals and then estimating time differences of arrival (TDOAs). The TDOA information is exploited to estimate the speakers’ activity, even in the presence of multiple speakers being simultaneously active. This speaker activity information serves as a guide for a spatial mixture model, on which basis the individual speaker’s signals are extracted via beamforming. Finally, the extracted signals are forwarded to a speech recognizer. Additionally, a novel initialization scheme for spatial mixture models based on the TDOA estimates is proposed. Experiments conducted on real recordings from the LibriWASN data set have shown that our proposed system is advantageous compared to a system using a spatial mixture model, which does not make use\r\nof external diarization information."}]},{"oa":"1","project":[{"name":"TRR 318 - C06: TRR 318 - Technisch unterstütztes Erklären von Stimmcharakteristika (Teilprojekt C06)","_id":"129","grant_number":"438445824"}],"citation":{"short":"F. Rautenberg, M. Kuhlmann, J. Ebbers, J. Wiechmann, F. Seebauer, P. Wagner, R. Haeb-Umbach, in: Fortschritte Der Akustik - DAGA 2023, 2023, pp. 1409–1412.","chicago":"Rautenberg, Frederik, Michael Kuhlmann, Janek Ebbers, Jana Wiechmann, Fritz Seebauer, Petra Wagner, and Reinhold Haeb-Umbach. “Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics.” In <i>Fortschritte Der Akustik - DAGA 2023</i>, 1409–12, 2023.","ieee":"F. Rautenberg <i>et al.</i>, “Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics,” in <i>Fortschritte der Akustik - DAGA 2023</i>, Hamburg, 2023, pp. 1409–1412.","apa":"Rautenberg, F., Kuhlmann, M., Ebbers, J., Wiechmann, J., Seebauer, F., Wagner, P., &#38; Haeb-Umbach, R. (2023). Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics. <i>Fortschritte Der Akustik - DAGA 2023</i>, 1409–1412.","bibtex":"@inproceedings{Rautenberg_Kuhlmann_Ebbers_Wiechmann_Seebauer_Wagner_Haeb-Umbach_2023, title={Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics}, booktitle={Fortschritte der Akustik - DAGA 2023}, author={Rautenberg, Frederik and Kuhlmann, Michael and Ebbers, Janek and Wiechmann, Jana and Seebauer, Fritz and Wagner, Petra and Haeb-Umbach, Reinhold}, year={2023}, pages={1409–1412} }","ama":"Rautenberg F, Kuhlmann M, Ebbers J, et al. Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics. In: <i>Fortschritte Der Akustik - DAGA 2023</i>. ; 2023:1409-1412.","mla":"Rautenberg, Frederik, et al. “Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics.” <i>Fortschritte Der Akustik - DAGA 2023</i>, 2023, pp. 1409–12."},"file_date_updated":"2024-02-29T16:15:12Z","ddc":["000"],"user_id":"72602","_id":"44849","page":"1409-1412","has_accepted_license":"1","conference":{"start_date":"2023-03-06","name":"DAGA 2023 - 49. Jahrestagung für Akustik","location":"Hamburg","end_date":"2023-03-09"},"status":"public","department":[{"_id":"54"},{"_id":"660"}],"type":"conference","date_created":"2023-05-15T08:48:54Z","file":[{"creator":"frra","date_created":"2024-02-29T16:15:12Z","file_size":289493,"access_level":"open_access","file_name":"Daga_2023_Rautenberg_Paper.pdf","date_updated":"2024-02-29T16:15:12Z","relation":"main_file","content_type":"application/pdf","file_id":"52221"}],"publication":"Fortschritte der Akustik - DAGA 2023","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://pub.dega-akustik.de/DAGA_2023/data/articles/000105.pdf","open_access":"1"}],"date_updated":"2024-02-29T17:05:16Z","publication_status":"published","author":[{"id":"72602","full_name":"Rautenberg, Frederik","first_name":"Frederik","last_name":"Rautenberg"},{"last_name":"Kuhlmann","first_name":"Michael","full_name":"Kuhlmann, Michael","id":"49871"},{"id":"34851","full_name":"Ebbers, Janek","first_name":"Janek","last_name":"Ebbers"},{"last_name":"Wiechmann","first_name":"Jana","full_name":"Wiechmann, Jana"},{"full_name":"Seebauer, Fritz","last_name":"Seebauer","first_name":"Fritz"},{"first_name":"Petra","last_name":"Wagner","full_name":"Wagner, Petra"},{"full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold","id":"242"}],"year":"2023","title":"Speech Disentanglement for Analysis and Modification of Acoustic and Perceptual Speaker Characteristics"},{"publication":"Speech Communication; 15th ITG Conference","citation":{"ieee":"M. Kuhlmann, A. T. Meise, F. Seebauer, P. Wagner, and R. Häb-Umbach, “Investigating Speaker Embedding Disentanglement on Natural Read Speech,” in <i>Speech Communication; 15th ITG Conference</i>, 2023, pp. 121–125.","apa":"Kuhlmann, M., Meise, A. T., Seebauer, F., Wagner, P., &#38; Häb-Umbach, R. (2023). Investigating Speaker Embedding Disentanglement on Natural Read Speech. <i>Speech Communication; 15th ITG Conference</i>, 121–125.","short":"M. Kuhlmann, A.T. Meise, F. Seebauer, P. Wagner, R. Häb-Umbach, in: Speech Communication; 15th ITG Conference, 2023, pp. 121–125.","chicago":"Kuhlmann, Michael, Adrian Tobias Meise, Fritz Seebauer, Petra Wagner, and Reinhold Häb-Umbach. “Investigating Speaker Embedding Disentanglement on Natural Read Speech.” In <i>Speech Communication; 15th ITG Conference</i>, 121–125, 2023.","mla":"Kuhlmann, Michael, et al. “Investigating Speaker Embedding Disentanglement on Natural Read Speech.” <i>Speech Communication; 15th ITG Conference</i>, 2023, pp. 121–125.","bibtex":"@inproceedings{Kuhlmann_Meise_Seebauer_Wagner_Häb-Umbach_2023, title={Investigating Speaker Embedding Disentanglement on Natural Read Speech}, booktitle={Speech Communication; 15th ITG Conference}, author={Kuhlmann, Michael and Meise, Adrian Tobias and Seebauer, Fritz and Wagner, Petra and Häb-Umbach, Reinhold}, year={2023}, pages={121–125} }","ama":"Kuhlmann M, Meise AT, Seebauer F, Wagner P, Häb-Umbach R. Investigating Speaker Embedding Disentanglement on Natural Read Speech. In: <i>Speech Communication; 15th ITG Conference</i>. ; 2023:121–125."},"project":[{"_id":"52","name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"date_created":"2024-11-14T09:45:03Z","type":"conference","department":[{"_id":"54"}],"year":"2023","status":"public","title":"Investigating Speaker Embedding Disentanglement on Natural Read Speech","author":[{"full_name":"Kuhlmann, Michael","first_name":"Michael","last_name":"Kuhlmann","id":"49871"},{"id":"79268","full_name":"Meise, Adrian Tobias","first_name":"Adrian Tobias","last_name":"Meise"},{"last_name":"Seebauer","first_name":"Fritz","full_name":"Seebauer, Fritz"},{"full_name":"Wagner, Petra","last_name":"Wagner","first_name":"Petra"},{"last_name":"Häb-Umbach","first_name":"Reinhold","full_name":"Häb-Umbach, Reinhold","id":"242"}],"date_updated":"2026-01-05T10:12:23Z","page":"121–125","_id":"57086","language":[{"iso":"eng"}],"user_id":"49871"},{"status":"public","has_accepted_license":"1","page":"36–40","_id":"49111","ddc":["000"],"user_id":"34851","file_date_updated":"2023-11-22T08:25:08Z","citation":{"ieee":"J. Ebbers, R. Haeb-Umbach, and R. Serizel, “Post-Processing Independent Evaluation of Sound Event Detection Systems,” in <i>Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)</i>, 2023, pp. 36–40.","apa":"Ebbers, J., Haeb-Umbach, R., &#38; Serizel, R. (2023). Post-Processing Independent Evaluation of Sound Event Detection Systems. <i>Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)</i>, 36–40.","chicago":"Ebbers, Janek, Reinhold Haeb-Umbach, and Romain Serizel. “Post-Processing Independent Evaluation of Sound Event Detection Systems.” In <i>Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)</i>, 36–40. Tampere, Finland, 2023.","short":"J. Ebbers, R. Haeb-Umbach, R. Serizel, in: Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023), Tampere, Finland, 2023, pp. 36–40.","mla":"Ebbers, Janek, et al. “Post-Processing Independent Evaluation of Sound Event Detection Systems.” <i>Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)</i>, 2023, pp. 36–40.","bibtex":"@inproceedings{Ebbers_Haeb-Umbach_Serizel_2023, place={Tampere, Finland}, title={Post-Processing Independent Evaluation of Sound Event Detection Systems}, booktitle={Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)}, author={Ebbers, Janek and Haeb-Umbach, Reinhold and Serizel, Romain}, year={2023}, pages={36–40} }","ama":"Ebbers J, Haeb-Umbach R, Serizel R. Post-Processing Independent Evaluation of Sound Event Detection Systems. In: <i>Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)</i>. ; 2023:36–40."},"quality_controlled":"1","project":[{"_id":"52","name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing"}],"place":"Tampere, Finland","year":"2023","title":"Post-Processing Independent Evaluation of Sound Event Detection Systems","author":[{"id":"34851","last_name":"Ebbers","first_name":"Janek","full_name":"Ebbers, Janek"},{"full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach","id":"242"},{"last_name":"Serizel","first_name":"Romain","full_name":"Serizel, Romain"}],"date_updated":"2024-11-15T20:34:18Z","language":[{"iso":"eng"}],"publication":"Proceedings of the 8th Detection and Classification of Acoustic Scenes and Events 2023 Workshop (DCASE2023)","abstract":[{"text":"Due to the high variation in the application requirements of sound event detection (SED) systems, it is not sufficient to evaluate systems only in a single operating mode. Therefore, the community recently adopted the polyphonic sound detection score (PSDS) as an evaluation metric, which is the normalized area under the PSD receiver operating characteristic (PSD-ROC). It summarizes the system performance over a range of operating modes resulting from varying the decision threshold that is used to translate the system output scores into a binary detection output. Hence, it provides a more complete picture of the overall system behavior and is less biased by specific threshold tuning. However, besides the decision threshold there is also the post-processing that can be changed to enter another operating mode. In this paper we propose the post-processing independent PSDS (piPSDS) as a generalization of the PSDS. Here, the post-processing independent PSD-ROC includes operating points from varying post-processings with varying decision thresholds. Thus, it summarizes even more operating modes of an SED system and allows for system comparison without the need of implementing a post-processing and without a bias due to different post-processings. While piPSDS can in principle combine different types of post-processing, we here, as a first step, present median filter independent PSDS (miPSDS) results for this year’s DCASE Challenge Task4a systems. Source code is publicly available in our sed_scores_eval package (https://github.com/fgnt/sed_scores_eval).","lang":"eng"}],"file":[{"date_created":"2023-11-22T08:25:08Z","creator":"ebbers","file_id":"49112","success":1,"content_type":"application/pdf","relation":"main_file","date_updated":"2023-11-22T08:25:08Z","file_name":"dcase2023_ebbers.pdf","access_level":"closed","file_size":221875}],"date_created":"2023-11-22T08:20:26Z","type":"conference","department":[{"_id":"54"}]},{"date_updated":"2024-11-15T06:54:55Z","title":"DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH","status":"public","year":"2023","conference":{"name":"International Congress of Phonetic Sciences (ICPhS)","start_date":"2023-08-07","location":"Prague","end_date":"2023-08-11"},"author":[{"full_name":"Seebauer, Fritz","first_name":"Fritz","last_name":"Seebauer"},{"id":"49871","full_name":"Kuhlmann, Michael","last_name":"Kuhlmann","first_name":"Michael"},{"id":"242","first_name":"Reinhold","last_name":"Häb-Umbach","full_name":"Häb-Umbach, Reinhold"},{"full_name":"Wagner, Petra","last_name":"Wagner","first_name":"Petra"}],"publication_identifier":{"isbn":["978-80-908 114-2-3"]},"user_id":"49871","_id":"57098","language":[{"iso":"eng"}],"publication":"Proceedings of the 20th International Congress of Phonetic Sciences","citation":{"ama":"Seebauer F, Kuhlmann M, Häb-Umbach R, Wagner P. DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH. In: <i>Proceedings of the 20th International Congress of Phonetic Sciences</i>. ; 2023.","bibtex":"@inproceedings{Seebauer_Kuhlmann_Häb-Umbach_Wagner_2023, title={DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH}, booktitle={Proceedings of the 20th International Congress of Phonetic Sciences}, author={Seebauer, Fritz and Kuhlmann, Michael and Häb-Umbach, Reinhold and Wagner, Petra}, year={2023} }","mla":"Seebauer, Fritz, et al. “DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH.” <i>Proceedings of the 20th International Congress of Phonetic Sciences</i>, 2023.","chicago":"Seebauer, Fritz, Michael Kuhlmann, Reinhold Häb-Umbach, and Petra Wagner. “DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH.” In <i>Proceedings of the 20th International Congress of Phonetic Sciences</i>, 2023.","short":"F. Seebauer, M. Kuhlmann, R. Häb-Umbach, P. Wagner, in: Proceedings of the 20th International Congress of Phonetic Sciences, 2023.","apa":"Seebauer, F., Kuhlmann, M., Häb-Umbach, R., &#38; Wagner, P. (2023). DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH. <i>Proceedings of the 20th International Congress of Phonetic Sciences</i>. International Congress of Phonetic Sciences (ICPhS), Prague.","ieee":"F. Seebauer, M. Kuhlmann, R. Häb-Umbach, and P. Wagner, “DISCERNING DIMENSIONS OF QUALITY FOR STATE OF THE ART SYNTHETIC SPEECH,” presented at the International Congress of Phonetic Sciences (ICPhS), Prague, 2023."},"type":"conference","department":[{"_id":"54"}],"date_created":"2024-11-15T06:49:27Z"},{"oa":"1","project":[{"name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"},{"name":"Automatische Transkription von Gesprächssituationen","_id":"508","grant_number":"448568305"}],"quality_controlled":"1","citation":{"ieee":"T. von Neumann, C. Boeddeker, K. Kinoshita, M. Delcroix, and R. Haeb-Umbach, “On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems,” 2023, doi: <a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">10.1109/icassp49357.2023.10094784</a>.","apa":"von Neumann, T., Boeddeker, C., Kinoshita, K., Delcroix, M., &#38; Haeb-Umbach, R. (2023). On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems. <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. <a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">https://doi.org/10.1109/icassp49357.2023.10094784</a>","short":"T. von Neumann, C. Boeddeker, K. Kinoshita, M. Delcroix, R. Haeb-Umbach, in: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2023.","chicago":"Neumann, Thilo von, Christoph Boeddeker, Keisuke Kinoshita, Marc Delcroix, and Reinhold Haeb-Umbach. “On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems.” In <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE, 2023. <a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">https://doi.org/10.1109/icassp49357.2023.10094784</a>.","mla":"von Neumann, Thilo, et al. “On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems.” <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, IEEE, 2023, doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">10.1109/icassp49357.2023.10094784</a>.","bibtex":"@inproceedings{von Neumann_Boeddeker_Kinoshita_Delcroix_Haeb-Umbach_2023, title={On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems}, DOI={<a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">10.1109/icassp49357.2023.10094784</a>}, booktitle={ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, publisher={IEEE}, author={von Neumann, Thilo and Boeddeker, Christoph and Kinoshita, Keisuke and Delcroix, Marc and Haeb-Umbach, Reinhold}, year={2023} }","ama":"von Neumann T, Boeddeker C, Kinoshita K, Delcroix M, Haeb-Umbach R. On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems. In: <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE; 2023. doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10094784\">10.1109/icassp49357.2023.10094784</a>"},"file_date_updated":"2023-10-19T07:41:56Z","user_id":"40767","ddc":["000"],"_id":"48281","publisher":"IEEE","has_accepted_license":"1","status":"public","department":[{"_id":"54"}],"type":"conference","keyword":["Word Error Rate","Meeting Recognition","Levenshtein Distance"],"date_created":"2023-10-19T07:38:31Z","file":[{"file_id":"48282","content_type":"application/pdf","relation":"main_file","date_updated":"2023-10-19T07:41:56Z","file_name":"ICASSP_2023_Meeting_Evaluation.pdf","access_level":"open_access","file_size":204994,"date_created":"2023-10-19T07:39:57Z","creator":"tvn"}],"abstract":[{"text":"\tWe propose a general framework to compute the word error rate (WER) of ASR systems that process recordings containing multiple speakers at their input and that produce multiple output word sequences (MIMO).\r\n\tSuch ASR systems are typically required, e.g., for meeting transcription.\r\n\tWe provide an efficient implementation based on a dynamic programming search in a multi-dimensional Levenshtein distance tensor under the constraint that a reference utterance must be matched consistently with one hypothesis output. \r\n\tThis also results in an efficient implementation of the ORC WER which previously suffered from exponential complexity.\r\n\tWe give an overview of commonly used WER definitions for multi-speaker scenarios and show that they are specializations of the above MIMO WER tuned to particular application scenarios. \r\n\tWe conclude with a  discussion of the pros and cons of the various WER definitions and a recommendation when to use which.","lang":"eng"}],"related_material":{"link":[{"relation":"software","url":"https://github.com/fgnt/meeteval"}]},"publication":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","doi":"10.1109/icassp49357.2023.10094784","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://ieeexplore.ieee.org/document/10094784"}],"publication_status":"published","date_updated":"2025-02-12T09:16:34Z","author":[{"full_name":"von Neumann, Thilo","orcid":"https://orcid.org/0000-0002-7717-8670","last_name":"von Neumann","first_name":"Thilo","id":"49870"},{"id":"40767","full_name":"Boeddeker, Christoph","last_name":"Boeddeker","first_name":"Christoph"},{"full_name":"Kinoshita, Keisuke","last_name":"Kinoshita","first_name":"Keisuke"},{"first_name":"Marc","last_name":"Delcroix","full_name":"Delcroix, Marc"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"title":"On Word Error Rate Definitions and Their Efficient Computation for Multi-Speaker Speech Recognition Systems","year":"2023"},{"conference":{"name":"CHiME 2023 Workshop on Speech Processing in Everyday Environments","location":"Dublin"},"status":"public","has_accepted_license":"1","_id":"48275","ddc":["000"],"user_id":"40767","citation":{"ieee":"T. von Neumann, C. Boeddeker, M. Delcroix, and R. Haeb-Umbach, “MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems,” presented at the CHiME 2023 Workshop on Speech Processing in Everyday Environments, Dublin, 2023.","mla":"von Neumann, Thilo, et al. “MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems.” <i>Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments</i>, 2023.","apa":"von Neumann, T., Boeddeker, C., Delcroix, M., &#38; Haeb-Umbach, R. (2023). MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems. <i>Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments</i>. CHiME 2023 Workshop on Speech Processing in Everyday Environments, Dublin.","bibtex":"@inproceedings{von Neumann_Boeddeker_Delcroix_Haeb-Umbach_2023, title={MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems}, booktitle={Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments}, author={von Neumann, Thilo and Boeddeker, Christoph and Delcroix, Marc and Haeb-Umbach, Reinhold}, year={2023} }","short":"T. von Neumann, C. Boeddeker, M. Delcroix, R. Haeb-Umbach, in: Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments, 2023.","ama":"von Neumann T, Boeddeker C, Delcroix M, Haeb-Umbach R. MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems. In: <i>Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments</i>. ; 2023.","chicago":"Neumann, Thilo von, Christoph Boeddeker, Marc Delcroix, and Reinhold Haeb-Umbach. “MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems.” In <i>Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments</i>, 2023."},"file_date_updated":"2023-10-19T07:19:59Z","project":[{"_id":"52","name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing"},{"name":"Automatische Transkription von Gesprächssituationen","_id":"508","grant_number":"448568305"}],"quality_controlled":"1","oa":"1","author":[{"id":"49870","last_name":"von Neumann","orcid":"https://orcid.org/0000-0002-7717-8670","first_name":"Thilo","full_name":"von Neumann, Thilo"},{"full_name":"Boeddeker, Christoph","first_name":"Christoph","last_name":"Boeddeker","id":"40767"},{"full_name":"Delcroix, Marc","first_name":"Marc","last_name":"Delcroix"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"title":"MeetEval: A Toolkit for Computation of Word Error Rates for Meeting Transcription Systems","year":"2023","date_updated":"2025-02-12T09:12:05Z","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://arxiv.org/abs/2307.11394","open_access":"1"}],"publication":"Proc. CHiME 2023 Workshop on Speech Processing in Everyday Environments","abstract":[{"lang":"eng","text":"MeetEval is an open-source toolkit to evaluate  all kinds of meeting transcription systems.\r\nIt provides a unified interface for the computation of commonly used Word Error Rates (WERs), specifically cpWER, ORC WER and MIMO WER along other WER definitions.\r\nWe extend the cpWER computation by a temporal constraint to ensure that only words are identified as correct when the temporal alignment is plausible.\r\nThis leads to a better quality of the matching of the hypothesis string to the reference string that more closely resembles the actual transcription quality, and a system is penalized if it provides poor time annotations.\r\nSince word-level timing information is often not available, we present a way to approximate exact word-level timings from segment-level timings (e.g., a sentence) and show that the approximation leads to a similar WER as a matching with exact word-level annotations.\r\nAt the same time, the time constraint leads to a speedup of the matching algorithm, which outweighs the additional overhead caused by processing the time stamps."}],"related_material":{"link":[{"relation":"software","url":"https://github.com/fgnt/meeteval"}]},"date_created":"2023-10-19T07:24:51Z","file":[{"date_created":"2023-10-19T07:19:59Z","creator":"tvn","content_type":"application/pdf","file_id":"48276","date_updated":"2023-10-19T07:19:59Z","relation":"main_file","access_level":"open_access","file_size":263744,"file_name":"Chime_7__MeetEval.pdf"}],"department":[{"_id":"54"}],"keyword":["Speech Recognition","Word Error Rate","Meeting Transcription"],"type":"conference"},{"author":[{"id":"44393","first_name":"Tobias","last_name":"Cord-Landwehr","full_name":"Cord-Landwehr, Tobias"},{"first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph","id":"40767"},{"last_name":"Zorilă","first_name":"Cătălin","full_name":"Zorilă, Cătălin"},{"first_name":"Rama","last_name":"Doddipatla","full_name":"Doddipatla, Rama"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","last_name":"Haeb-Umbach","first_name":"Reinhold"}],"title":"Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization","year":"2023","date_updated":"2025-02-12T09:14:45Z","publication_status":"published","language":[{"iso":"eng"}],"doi":"10.1109/icassp49357.2023.10095370","publication":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","date_created":"2023-09-19T14:01:20Z","file":[{"date_created":"2023-11-15T14:56:18Z","creator":"cord","content_type":"application/pdf","file_id":"48932","date_updated":"2023-11-15T14:56:18Z","relation":"main_file","access_level":"open_access","file_size":246306,"file_name":"teacher_student_embeddings.pdf"}],"department":[{"_id":"54"}],"type":"conference","conference":{"location":"Rhodes","name":"2023 IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP)"},"status":"public","has_accepted_license":"1","publisher":"IEEE","_id":"47128","ddc":["000"],"user_id":"40767","citation":{"bibtex":"@inproceedings{Cord-Landwehr_Boeddeker_Zorilă_Doddipatla_Haeb-Umbach_2023, title={Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization}, DOI={<a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">10.1109/icassp49357.2023.10095370</a>}, booktitle={ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, publisher={IEEE}, author={Cord-Landwehr, Tobias and Boeddeker, Christoph and Zorilă, Cătălin and Doddipatla, Rama and Haeb-Umbach, Reinhold}, year={2023} }","ama":"Cord-Landwehr T, Boeddeker C, Zorilă C, Doddipatla R, Haeb-Umbach R. Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization. In: <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE; 2023. doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">10.1109/icassp49357.2023.10095370</a>","short":"T. Cord-Landwehr, C. Boeddeker, C. Zorilă, R. Doddipatla, R. Haeb-Umbach, in: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, 2023.","chicago":"Cord-Landwehr, Tobias, Christoph Boeddeker, Cătălin Zorilă, Rama Doddipatla, and Reinhold Haeb-Umbach. “Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization.” In <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. IEEE, 2023. <a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">https://doi.org/10.1109/icassp49357.2023.10095370</a>.","ieee":"T. Cord-Landwehr, C. Boeddeker, C. Zorilă, R. Doddipatla, and R. Haeb-Umbach, “Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization,” presented at the 2023 IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Rhodes, 2023, doi: <a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">10.1109/icassp49357.2023.10095370</a>.","mla":"Cord-Landwehr, Tobias, et al. “Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization.” <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>, IEEE, 2023, doi:<a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">10.1109/icassp49357.2023.10095370</a>.","apa":"Cord-Landwehr, T., Boeddeker, C., Zorilă, C., Doddipatla, R., &#38; Haeb-Umbach, R. (2023). Frame-Wise and Overlap-Robust Speaker Embeddings for Meeting Diarization. <i>ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</i>. 2023 IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Rhodes. <a href=\"https://doi.org/10.1109/icassp49357.2023.10095370\">https://doi.org/10.1109/icassp49357.2023.10095370</a>"},"file_date_updated":"2023-11-15T14:56:18Z","project":[{"name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"},{"name":"Automatische Transkription von Gesprächssituationen","_id":"508","grant_number":"448568305"}],"oa":"1"},{"ddc":["000"],"user_id":"40767","publisher":"ISCA","_id":"47129","has_accepted_license":"1","status":"public","oa":"1","project":[{"name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"},{"_id":"508","grant_number":"448568305","name":"Automatische Transkription von Gesprächssituationen"}],"citation":{"apa":"Cord-Landwehr, T., Boeddeker, C., Zorilă, C., Doddipatla, R., &#38; Haeb-Umbach, R. (2023). A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures. <i>INTERSPEECH 2023</i>. <a href=\"https://doi.org/10.21437/interspeech.2023-1379\">https://doi.org/10.21437/interspeech.2023-1379</a>","ieee":"T. Cord-Landwehr, C. Boeddeker, C. Zorilă, R. Doddipatla, and R. Haeb-Umbach, “A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures,” 2023, doi: <a href=\"https://doi.org/10.21437/interspeech.2023-1379\">10.21437/interspeech.2023-1379</a>.","short":"T. Cord-Landwehr, C. Boeddeker, C. Zorilă, R. Doddipatla, R. Haeb-Umbach, in: INTERSPEECH 2023, ISCA, 2023.","chicago":"Cord-Landwehr, Tobias, Christoph Boeddeker, Cătălin Zorilă, Rama Doddipatla, and Reinhold Haeb-Umbach. “A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures.” In <i>INTERSPEECH 2023</i>. ISCA, 2023. <a href=\"https://doi.org/10.21437/interspeech.2023-1379\">https://doi.org/10.21437/interspeech.2023-1379</a>.","mla":"Cord-Landwehr, Tobias, et al. “A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures.” <i>INTERSPEECH 2023</i>, ISCA, 2023, doi:<a href=\"https://doi.org/10.21437/interspeech.2023-1379\">10.21437/interspeech.2023-1379</a>.","ama":"Cord-Landwehr T, Boeddeker C, Zorilă C, Doddipatla R, Haeb-Umbach R. A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures. In: <i>INTERSPEECH 2023</i>. ISCA; 2023. doi:<a href=\"https://doi.org/10.21437/interspeech.2023-1379\">10.21437/interspeech.2023-1379</a>","bibtex":"@inproceedings{Cord-Landwehr_Boeddeker_Zorilă_Doddipatla_Haeb-Umbach_2023, title={A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures}, DOI={<a href=\"https://doi.org/10.21437/interspeech.2023-1379\">10.21437/interspeech.2023-1379</a>}, booktitle={INTERSPEECH 2023}, publisher={ISCA}, author={Cord-Landwehr, Tobias and Boeddeker, Christoph and Zorilă, Cătălin and Doddipatla, Rama and Haeb-Umbach, Reinhold}, year={2023} }"},"file_date_updated":"2023-11-15T15:00:02Z","doi":"10.21437/interspeech.2023-1379","language":[{"iso":"eng"}],"date_updated":"2025-02-12T09:15:28Z","publication_status":"published","author":[{"full_name":"Cord-Landwehr, Tobias","last_name":"Cord-Landwehr","first_name":"Tobias","id":"44393"},{"full_name":"Boeddeker, Christoph","first_name":"Christoph","last_name":"Boeddeker","id":"40767"},{"full_name":"Zorilă, Cătălin","first_name":"Cătălin","last_name":"Zorilă"},{"first_name":"Rama","last_name":"Doddipatla","full_name":"Doddipatla, Rama"},{"first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"year":"2023","title":"A Teacher-Student Approach for Extracting Informative Speaker Embeddings From Speech Mixtures","department":[{"_id":"54"}],"type":"conference","date_created":"2023-09-19T14:34:37Z","file":[{"file_name":"multispeaker_embeddings.pdf","access_level":"open_access","file_size":303203,"relation":"main_file","date_updated":"2023-11-15T15:00:02Z","file_id":"48933","content_type":"application/pdf","creator":"cord","date_created":"2023-11-15T15:00:02Z"}],"publication":"INTERSPEECH 2023"},{"user_id":"40767","doi":"10.21437/chime.2023-10","_id":"54439","publisher":"ISCA","language":[{"iso":"eng"}],"main_file_link":[{"url":"https://www.isca-archive.org/chime_2023/boeddeker23_chime.pdf","open_access":"1"}],"publication_status":"published","date_updated":"2025-02-12T09:16:13Z","author":[{"id":"40767","first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph"},{"full_name":"Cord-Landwehr, Tobias","last_name":"Cord-Landwehr","first_name":"Tobias","id":"44393"},{"orcid":"https://orcid.org/0000-0002-7717-8670","last_name":"von Neumann","first_name":"Thilo","full_name":"von Neumann, Thilo","id":"49870"},{"id":"242","full_name":"Haeb-Umbach, Reinhold","first_name":"Reinhold","last_name":"Haeb-Umbach"}],"year":"2023","status":"public","title":"Multi-stage diarization refinement for the CHiME-7 DASR scenario","department":[{"_id":"54"}],"oa":"1","type":"conference","date_created":"2024-05-23T15:16:15Z","project":[{"_id":"52","name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing"},{"_id":"508","grant_number":"448568305","name":"Automatische Transkription von Gesprächssituationen"}],"citation":{"ieee":"C. Boeddeker, T. Cord-Landwehr, T. von Neumann, and R. Haeb-Umbach, “Multi-stage diarization refinement for the CHiME-7 DASR scenario,” 2023, doi: <a href=\"https://doi.org/10.21437/chime.2023-10\">10.21437/chime.2023-10</a>.","apa":"Boeddeker, C., Cord-Landwehr, T., von Neumann, T., &#38; Haeb-Umbach, R. (2023). Multi-stage diarization refinement for the CHiME-7 DASR scenario. <i>7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)</i>. <a href=\"https://doi.org/10.21437/chime.2023-10\">https://doi.org/10.21437/chime.2023-10</a>","chicago":"Boeddeker, Christoph, Tobias Cord-Landwehr, Thilo von Neumann, and Reinhold Haeb-Umbach. “Multi-Stage Diarization Refinement for the CHiME-7 DASR Scenario.” In <i>7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)</i>. ISCA, 2023. <a href=\"https://doi.org/10.21437/chime.2023-10\">https://doi.org/10.21437/chime.2023-10</a>.","short":"C. Boeddeker, T. Cord-Landwehr, T. von Neumann, R. Haeb-Umbach, in: 7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023), ISCA, 2023.","mla":"Boeddeker, Christoph, et al. “Multi-Stage Diarization Refinement for the CHiME-7 DASR Scenario.” <i>7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)</i>, ISCA, 2023, doi:<a href=\"https://doi.org/10.21437/chime.2023-10\">10.21437/chime.2023-10</a>.","bibtex":"@inproceedings{Boeddeker_Cord-Landwehr_von Neumann_Haeb-Umbach_2023, title={Multi-stage diarization refinement for the CHiME-7 DASR scenario}, DOI={<a href=\"https://doi.org/10.21437/chime.2023-10\">10.21437/chime.2023-10</a>}, booktitle={7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)}, publisher={ISCA}, author={Boeddeker, Christoph and Cord-Landwehr, Tobias and von Neumann, Thilo and Haeb-Umbach, Reinhold}, year={2023} }","ama":"Boeddeker C, Cord-Landwehr T, von Neumann T, Haeb-Umbach R. Multi-stage diarization refinement for the CHiME-7 DASR scenario. In: <i>7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)</i>. ISCA; 2023. doi:<a href=\"https://doi.org/10.21437/chime.2023-10\">10.21437/chime.2023-10</a>"},"publication":"7th International Workshop on Speech Processing in Everyday Environments (CHiME 2023)"},{"main_file_link":[{"url":"https://www.isca-archive.org/interspeech_2023/berger23_interspeech.pdf","open_access":"1"}],"_id":"48390","language":[{"iso":"eng"}],"publisher":"ISCA","doi":"10.21437/interspeech.2023-1815","user_id":"40767","status":"public","year":"2023","title":"Mixture Encoder for Joint Speech Separation and Recognition","author":[{"last_name":"Berger","first_name":"Simon","full_name":"Berger, Simon"},{"last_name":"Vieting","first_name":"Peter","full_name":"Vieting, Peter"},{"id":"40767","first_name":"Christoph","last_name":"Boeddeker","full_name":"Boeddeker, Christoph"},{"full_name":"Schlüter, Ralf","last_name":"Schlüter","first_name":"Ralf"},{"last_name":"Haeb-Umbach","first_name":"Reinhold","full_name":"Haeb-Umbach, Reinhold","id":"242"}],"date_updated":"2025-02-12T09:11:30Z","publication_status":"published","date_created":"2023-10-23T15:06:39Z","type":"conference","oa":"1","department":[{"_id":"54"}],"publication":"INTERSPEECH 2023","citation":{"ama":"Berger S, Vieting P, Boeddeker C, Schlüter R, Haeb-Umbach R. Mixture Encoder for Joint Speech Separation and Recognition. In: <i>INTERSPEECH 2023</i>. ISCA; 2023. doi:<a href=\"https://doi.org/10.21437/interspeech.2023-1815\">10.21437/interspeech.2023-1815</a>","bibtex":"@inproceedings{Berger_Vieting_Boeddeker_Schlüter_Haeb-Umbach_2023, title={Mixture Encoder for Joint Speech Separation and Recognition}, DOI={<a href=\"https://doi.org/10.21437/interspeech.2023-1815\">10.21437/interspeech.2023-1815</a>}, booktitle={INTERSPEECH 2023}, publisher={ISCA}, author={Berger, Simon and Vieting, Peter and Boeddeker, Christoph and Schlüter, Ralf and Haeb-Umbach, Reinhold}, year={2023} }","mla":"Berger, Simon, et al. “Mixture Encoder for Joint Speech Separation and Recognition.” <i>INTERSPEECH 2023</i>, ISCA, 2023, doi:<a href=\"https://doi.org/10.21437/interspeech.2023-1815\">10.21437/interspeech.2023-1815</a>.","short":"S. Berger, P. Vieting, C. Boeddeker, R. Schlüter, R. Haeb-Umbach, in: INTERSPEECH 2023, ISCA, 2023.","chicago":"Berger, Simon, Peter Vieting, Christoph Boeddeker, Ralf Schlüter, and Reinhold Haeb-Umbach. “Mixture Encoder for Joint Speech Separation and Recognition.” In <i>INTERSPEECH 2023</i>. ISCA, 2023. <a href=\"https://doi.org/10.21437/interspeech.2023-1815\">https://doi.org/10.21437/interspeech.2023-1815</a>.","apa":"Berger, S., Vieting, P., Boeddeker, C., Schlüter, R., &#38; Haeb-Umbach, R. (2023). Mixture Encoder for Joint Speech Separation and Recognition. <i>INTERSPEECH 2023</i>. <a href=\"https://doi.org/10.21437/interspeech.2023-1815\">https://doi.org/10.21437/interspeech.2023-1815</a>","ieee":"S. Berger, P. Vieting, C. Boeddeker, R. Schlüter, and R. Haeb-Umbach, “Mixture Encoder for Joint Speech Separation and Recognition,” 2023, doi: <a href=\"https://doi.org/10.21437/interspeech.2023-1815\">10.21437/interspeech.2023-1815</a>."},"project":[{"name":"Automatische Transkription von Gesprächssituationen","grant_number":"448568305","_id":"508"}]},{"language":[{"iso":"eng"}],"doi":"10.1109/TASLP.2022.3209942","publication_identifier":{"issn":["Print ISSN: 2329-9290 Electronic ISSN: 2329-9304"]},"author":[{"first_name":"Wangyou","last_name":"Zhang","full_name":"Zhang, Wangyou"},{"full_name":"Chang, Xuankai","last_name":"Chang","first_name":"Xuankai"},{"last_name":"Boeddeker","first_name":"Christoph","full_name":"Boeddeker, Christoph","id":"40767"},{"full_name":"Nakatani, Tomohiro","last_name":"Nakatani","first_name":"Tomohiro"},{"full_name":"Watanabe, Shinji","first_name":"Shinji","last_name":"Watanabe"},{"full_name":"Qian, Yanmin","first_name":"Yanmin","last_name":"Qian"}],"year":"2022","title":"End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party","date_updated":"2022-12-05T12:35:31Z","publication_status":"published","date_created":"2022-10-11T07:27:51Z","file":[{"access_level":"open_access","file_size":6167931,"file_name":"End-to-End_Dereverberation_Beamforming_and_Speech_Recognition_in_A_Cocktail_Party.pdf","date_updated":"2022-10-11T07:23:13Z","relation":"main_file","content_type":"application/pdf","file_id":"33674","creator":"huesera","date_created":"2022-10-11T07:23:13Z"}],"department":[{"_id":"54"}],"type":"journal_article","publication":"IEEE/ACM Transactions on Audio, Speech, and Language Processing","related_material":{"link":[{"relation":"confirmation","url":"https://ieeexplore.ieee.org/abstract/document/9904314"}]},"abstract":[{"lang":"eng","text":"Far-field multi-speaker automatic speech recognition (ASR) has drawn increasing attention in recent years. Most existing methods feature a signal processing frontend and an ASR backend. In realistic scenarios, these modules are usually trained separately or progressively, which suffers from either inter-module mismatch or a complicated training process. In this paper, we propose an end-to-end multi-channel model that jointly optimizes the speech enhancement (including speech dereverberation, denoising, and separation) frontend and the ASR backend as a single system. To the best of our knowledge, this is the first work that proposes to optimize dereverberation, beamforming, and multi-speaker ASR in a fully end-to-end manner. The frontend module consists of a weighted prediction error (WPE) based submodule for dereverberation and a neural beamformer for denoising and speech separation. For the backend, we adopt a widely used end-to-end (E2E) ASR architecture. It is worth noting that the entire model is differentiable and can be optimized in a fully end-to-end manner using only the ASR criterion, without the need of parallel signal-level labels. We evaluate the proposed model on several multi-speaker benchmark datasets, and experimental results show that the fully E2E ASR model can achieve competitive performance on both noisy and reverberant conditions, with over 30% relative word error rate (WER) reduction over the single-channel baseline systems."}],"_id":"33669","ddc":["000"],"user_id":"40767","status":"public","has_accepted_license":"1","oa":"1","citation":{"mla":"Zhang, Wangyou, et al. “End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, 2022, doi:<a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">10.1109/TASLP.2022.3209942</a>.","bibtex":"@article{Zhang_Chang_Boeddeker_Nakatani_Watanabe_Qian_2022, title={End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party}, DOI={<a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">10.1109/TASLP.2022.3209942</a>}, journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing}, author={Zhang, Wangyou and Chang, Xuankai and Boeddeker, Christoph and Nakatani, Tomohiro and Watanabe, Shinji and Qian, Yanmin}, year={2022} }","ama":"Zhang W, Chang X, Boeddeker C, Nakatani T, Watanabe S, Qian Y. End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>. Published online 2022. doi:<a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">10.1109/TASLP.2022.3209942</a>","ieee":"W. Zhang, X. Chang, C. Boeddeker, T. Nakatani, S. Watanabe, and Y. Qian, “End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party,” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, 2022, doi: <a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">10.1109/TASLP.2022.3209942</a>.","apa":"Zhang, W., Chang, X., Boeddeker, C., Nakatani, T., Watanabe, S., &#38; Qian, Y. (2022). End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party. <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>. <a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">https://doi.org/10.1109/TASLP.2022.3209942</a>","chicago":"Zhang, Wangyou, Xuankai Chang, Christoph Boeddeker, Tomohiro Nakatani, Shinji Watanabe, and Yanmin Qian. “End-to-End Dereverberation, Beamforming, and Speech Recognition in A Cocktail Party.” <i>IEEE/ACM Transactions on Audio, Speech, and Language Processing</i>, 2022. <a href=\"https://doi.org/10.1109/TASLP.2022.3209942\">https://doi.org/10.1109/TASLP.2022.3209942</a>.","short":"W. Zhang, X. Chang, C. Boeddeker, T. Nakatani, S. Watanabe, Y. Qian, IEEE/ACM Transactions on Audio, Speech, and Language Processing (2022)."},"file_date_updated":"2022-10-11T07:23:13Z"},{"title":"Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels","year":"2022","author":[{"id":"27643","last_name":"Heitkämper","first_name":"Jens","full_name":"Heitkämper, Jens"},{"id":"460","last_name":"Schmalenstroeer","first_name":"Joerg","full_name":"Schmalenstroeer, Joerg"},{"id":"242","first_name":"Reinhold","last_name":"Haeb-Umbach","full_name":"Haeb-Umbach, Reinhold"}],"date_updated":"2023-10-26T08:15:57Z","publication_status":"accepted","language":[{"iso":"eng"}],"publication":"Proceedings of the 30th European Signal Processing Conference (EUSIPCO)","abstract":[{"lang":"eng","text":"The intelligibility of demodulated audio signals from analog high frequency transmissions, e.g., using single-sideband\r\n(SSB) modulation, can be severely degraded by channel distortions and/or a mismatch between modulation and demodulation carrier frequency. In this work a neural network (NN)-based approach for carrier frequency offset (CFO) estimation from demodulated SSB signals is proposed, whereby a task specific architecture is presented. Additionally, a simulation framework for SSB signals is introduced and utilized for training the NNs. The CFO estimator is combined with a speech enhancement network to investigate its influence on the enhancement performance. The NN-based system is compared to a recently proposed pitch tracking based approach on publicly available data from real high frequency transmissions. Experiments show that the NN exhibits good CFO estimation properties and results in significant improvements in speech intelligibility, especially when combined with a noise reduction network."}],"file":[{"access_level":"closed","file_size":1231379,"file_name":"cfo.pdf","date_updated":"2022-09-22T10:48:31Z","relation":"main_file","success":1,"content_type":"application/pdf","file_id":"33472","creator":"jensheit","date_created":"2022-09-22T10:48:31Z"}],"date_created":"2022-09-22T10:56:13Z","type":"conference","department":[{"_id":"54"}],"status":"public","conference":{"location":"Belgrad","start_date":"2022-08-29","name":"30th European Signal Processing Conference (EUSIPCO)","end_date":"2022-09-02"},"has_accepted_license":"1","_id":"33471","ddc":["000"],"user_id":"460","file_date_updated":"2022-09-22T10:48:31Z","citation":{"ama":"Heitkämper J, Schmalenstroeer J, Haeb-Umbach R. Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels. In: <i>Proceedings of the 30th European Signal Processing Conference (EUSIPCO)</i>.","bibtex":"@inproceedings{Heitkämper_Schmalenstroeer_Haeb-Umbach, place={Belgrad}, title={Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels}, booktitle={Proceedings of the 30th European Signal Processing Conference (EUSIPCO)}, author={Heitkämper, Jens and Schmalenstroeer, Joerg and Haeb-Umbach, Reinhold} }","mla":"Heitkämper, Jens, et al. “Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels.” <i>Proceedings of the 30th European Signal Processing Conference (EUSIPCO)</i>.","chicago":"Heitkämper, Jens, Joerg Schmalenstroeer, and Reinhold Haeb-Umbach. “Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels.” In <i>Proceedings of the 30th European Signal Processing Conference (EUSIPCO)</i>. Belgrad, n.d.","short":"J. Heitkämper, J. Schmalenstroeer, R. Haeb-Umbach, in: Proceedings of the 30th European Signal Processing Conference (EUSIPCO), Belgrad, n.d.","apa":"Heitkämper, J., Schmalenstroeer, J., &#38; Haeb-Umbach, R. (n.d.). Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels. <i>Proceedings of the 30th European Signal Processing Conference (EUSIPCO)</i>. 30th European Signal Processing Conference (EUSIPCO), Belgrad.","ieee":"J. Heitkämper, J. Schmalenstroeer, and R. Haeb-Umbach, “Neural Network Based Carrier Frequency Offset Estimation From Speech Transmitted Over High Frequency Channels,” presented at the 30th European Signal Processing Conference (EUSIPCO), Belgrad."},"quality_controlled":"1","project":[{"name":"PC2: Computing Resources Provided by the Paderborn Center for Parallel Computing","_id":"52"}],"place":"Belgrad"},{"date_updated":"2023-10-26T08:16:07Z","publication_status":"published","author":[{"last_name":"Afifi","first_name":"Haitham","full_name":"Afifi, Haitham"},{"full_name":"Karl, Holger","first_name":"Holger","last_name":"Karl"},{"id":"44006","full_name":"Gburrek, Tobias","last_name":"Gburrek","first_name":"Tobias"},{"id":"460","full_name":"Schmalenstroeer, Joerg","last_name":"Schmalenstroeer","first_name":"Joerg"}],"year":"2022","title":"Data-driven Time Synchronization in Wireless Multimedia Networks","status":"public","doi":"10.1109/iwcmc55113.2022.9824980","user_id":"460","_id":"33806","publisher":"IEEE","language":[{"iso":"eng"}],"quality_controlled":"1","citation":{"bibtex":"@inproceedings{Afifi_Karl_Gburrek_Schmalenstroeer_2022, title={Data-driven Time Synchronization in Wireless Multimedia Networks}, DOI={<a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">10.1109/iwcmc55113.2022.9824980</a>}, booktitle={2022 International Wireless Communications and Mobile Computing (IWCMC)}, publisher={IEEE}, author={Afifi, Haitham and Karl, Holger and Gburrek, Tobias and Schmalenstroeer, Joerg}, year={2022} }","ama":"Afifi H, Karl H, Gburrek T, Schmalenstroeer J. Data-driven Time Synchronization in Wireless Multimedia Networks. In: <i>2022 International Wireless Communications and Mobile Computing (IWCMC)</i>. IEEE; 2022. doi:<a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">10.1109/iwcmc55113.2022.9824980</a>","mla":"Afifi, Haitham, et al. “Data-Driven Time Synchronization in Wireless Multimedia Networks.” <i>2022 International Wireless Communications and Mobile Computing (IWCMC)</i>, IEEE, 2022, doi:<a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">10.1109/iwcmc55113.2022.9824980</a>.","chicago":"Afifi, Haitham, Holger Karl, Tobias Gburrek, and Joerg Schmalenstroeer. “Data-Driven Time Synchronization in Wireless Multimedia Networks.” In <i>2022 International Wireless Communications and Mobile Computing (IWCMC)</i>. IEEE, 2022. <a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">https://doi.org/10.1109/iwcmc55113.2022.9824980</a>.","short":"H. Afifi, H. Karl, T. Gburrek, J. Schmalenstroeer, in: 2022 International Wireless Communications and Mobile Computing (IWCMC), IEEE, 2022.","ieee":"H. Afifi, H. Karl, T. Gburrek, and J. Schmalenstroeer, “Data-driven Time Synchronization in Wireless Multimedia Networks,” 2022, doi: <a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">10.1109/iwcmc55113.2022.9824980</a>.","apa":"Afifi, H., Karl, H., Gburrek, T., &#38; Schmalenstroeer, J. (2022). Data-driven Time Synchronization in Wireless Multimedia Networks. <i>2022 International Wireless Communications and Mobile Computing (IWCMC)</i>. <a href=\"https://doi.org/10.1109/iwcmc55113.2022.9824980\">https://doi.org/10.1109/iwcmc55113.2022.9824980</a>"},"publication":"2022 International Wireless Communications and Mobile Computing (IWCMC)","department":[{"_id":"54"}],"type":"conference","date_created":"2022-10-18T09:24:17Z"}]
