@inbook{61323,
  author       = {{Wrede, Britta and Buschmeier, Hendrik and Rohlfing, Katharina Justine and Booshehri, Meisam and Grimminger, Angela}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina J. and Främling, Kary and Alpsancar, Suzana and Thommes, Kirsten and Lim, Brian Y.}},
  pages        = {{227--245}},
  publisher    = {{Springer}},
  title        = {{{Incremental communication}}},
  doi          = {{10.1007/978-981-96-5290-7_12}},
  year         = {{2026}},
}

@inbook{61321,
  author       = {{Grimminger, Angela and Buschmeier, Hendrik}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina J. and Främling, Kary and Alpsancar, Suzana and Thommes, Kirsten and Lim, Brian Y.}},
  pages        = {{351--365}},
  publisher    = {{Springer}},
  title        = {{{Theoretical aspects of multimodal processing}}},
  doi          = {{10.1007/978-981-96-5290-7_18}},
  year         = {{2026}},
}

@inbook{61322,
  author       = {{Lazarov, Stefan Teodorov and Tchappi, Igor and Grimminger, Angela}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina J. and Främling, Kary and Alpsancar, Suzana and Thommes, Kirsten and Lim, Brian Y.}},
  pages        = {{367--390}},
  publisher    = {{Springer}},
  title        = {{{Characteristics of nonverbal behavior}}},
  doi          = {{10.1007/978-981-96-5290-7_19}},
  year         = {{2026}},
}

@inbook{61324,
  author       = {{Wagner, Petra and Kopp, Stefan}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina J. and Främling, Kary and Alpsancar, Suzana and Thommes, Kirsten and Lim, Brian Y.}},
  pages        = {{433--446}},
  publisher    = {{Springer}},
  title        = {{{Timing and synchronization of multimodal signals in explanations}}},
  doi          = {{10.1007/978-981-96-5290-7_22}},
  year         = {{2026}},
}

@inbook{61112,
  author       = {{Rohlfing, Katharina J. and Vollmer, Anna-Lisa and Grimminger, Angela}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina and Främling, Kary and Thommes, Kirsten and Alpsancar, Suzana and Lim, Brian Y.}},
  publisher    = {{Springer}},
  title        = {{{Practices: How to establish an explaining practice}}},
  doi          = {{10.1007/978-981-96-5290-7_5}},
  year         = {{2026}},
}

@inbook{61325,
  author       = {{Vollmer, Anna-Lisa and Buhl, Heike M. and Alami, Rachid and Främling, Kary and Grimminger, Angela and Booshehri, Meisam and Ngonga Ngomo, Axel-Cyrille}},
  booktitle    = {{Social Explainable AI}},
  editor       = {{Rohlfing, Katharina J. and Främling, Kary and Lim, Brian and Alpsancar, Suzana and Thommes, Kirsten}},
  pages        = {{39--53}},
  publisher    = {{Springer}},
  title        = {{{Components of an explanation for co-constructive sXAI}}},
  doi          = {{10.1007/978-981-96-5290-7_3}},
  year         = {{2026}},
}

@unpublished{61151,
  abstract     = {{In this paper, we discuss the application of retrospective video recall for the assessment of cognitive processes in explanatory interactions, such as understanding and mental models. Our purpose is to reflect on the benefits and limitations of video recall compared to another self-report method, ‘thinking-aloud’. To do so, we reveal empirical results from the application of video recall in three interdisciplinary research projects that applied the method for the qualitative and quantitative assessment of cognitive and behavioral phenomena in everyday explanations. In all three projects, video recall was applied as a post-hoc procedure following the recording of dyadic face-to-face explanations of board games. The design of the video recall procedure differed between individual projects because they pursued different research objectives – that is the investigation of (1) an interlocutor's multimodal signals of understanding, (2) the change in assumptions about an interlocutor's dispositional and situational knowledge, and (3) the differentiated assessment of an interlocutor's developing understanding of domain knowledge aspects by distinguishing between mechanistic and functional explanatory stances. By discussing the benefits and the limitations of each procedure, this article provides critical reflections on video recall as a versatile research method applied for the analysis of human multimodal behavior in interaction and cognitive processing.}},
  author       = {{Lazarov, Stefan Teodorov and Schaffer, Michael and Gladow, Viviane and Buschmeier, Hendrik and Buhl, Heike M. and Grimminger, Angela}},
  pages        = {{29}},
  title        = {{{Retrospective video recall for analyzing cognitive processes in naturalistic explanations}}},
  year         = {{2026}},
}

@inproceedings{64914,
  abstract     = {{We investigate how verbal and nonverbal linguistic features, exhibited by speakers and listeners in dialogue, can contribute to predicting the listener's state of understanding in explanatory interactions on a moment-by-moment basis. Specifically, we examine three linguistic cues related to cognitive load and hypothesised to correlate with listener understanding: the information value (operationalised with surprisal) and syntactic complexity of the speaker's utterances, and the variation in the listener's interactive gaze behaviour. Based on statistical analyses of the MUNDEX corpus of face-to-face dialogic board game explanations, we find that individual cues vary with the listener's level of understanding. Listener states (‘Understanding’, ‘Partial Understanding’, ‘Non-Understanding’ and ‘Misunderstanding’) were self-annotated by the listeners using a retrospective video-recall method. The results of a subsequent classification experiment, involving two off-the-shelf classifiers and a fine-tuned German BERT-based multimodal classifier, demonstrate that prediction of these four states of understanding is generally possible and improves when the three linguistic cues are considered alongside textual features.}},
  author       = {{Wang, Yu and Türk, Olcay and Grimminger, Angela and Buschmeier, Hendrik}},
  booktitle    = {{Proceedings of the 15th Language Resources and Evaluation Conference}},
  location     = {{Palma, Mallorca, Spain}},
  pages        = {{11368--11378}},
  publisher    = {{ELRA}},
  title        = {{{Predicting states of understanding in explanatory interactions using cognitive load-related linguistic cues}}},
  doi          = {{10.63317/4tsmsshhd3ad}},
  year         = {{2026}},
}

@article{65565,
  abstract     = {{<jats:title>Abstract</jats:title>
                  <jats:p>Gaze behavior, being continuously accessible to interlocutors in face-to-face interactions, serves as a cue for managing turn-taking, regulating the duration of topical sequences, and supporting cognitive processing in various everyday conversational contexts. The present study seeks to enhance the understanding of the relation between two forms of interactive gaze behavior – gaze aversions and mutual gaze – and the topical development in the explanatory discourse. To do so, we analyzed 24 dyadic board game explanations in which one explainer subsequently explained a board game to three different explainees while the board game was physically absent from the shared space. The main objective of the present study was to investigate the relation of gaze aversions and mutual gaze to the topical development of explanations. For this, based on previous research (Lazarov et al., 2024; Rossano, 2012) we hypothesized that (1) gaze aversions are more likely to be associated with topic changes than topic continuations, and that (2) mutual gaze is more likely to be associated with topic continuations than topic changes. In addition, we explored how the two forms of gaze behavior are related to the interlocutor who initiates a topic change or continuation. Our proportional analysis using a Generalized linear mixed effects model revealed that gaze aversions are related to topic changes initiated by both interlocutors. In contrast, the analysis did not reveal a significant relation between mutual gaze and topic continuations, which could be explained by the feedback elicitation function of mutual gaze at the end of speakers’ utterances (Bavelas et al., 2002; Brône et al., 2017; Kendon, 1967) while monitoring the addressees’ understanding (Clark &amp; Krych, 2004) and the complexity of the analyzed fixed and random effects.</jats:p>}},
  author       = {{Lazarov, Stefan Teodorov and Grimminger, Angela}},
  issn         = {{0191-5886}},
  journal      = {{Journal of Nonverbal Behavior}},
  publisher    = {{Springer Science and Business Media LLC}},
  title        = {{{How are gaze aversions and mutual gaze related to the topical development of dyadic explanatory interactions?}}},
  doi          = {{10.1007/s10919-026-00512-8}},
  year         = {{2026}},
}

@inproceedings{61444,
  abstract     = {{Backchannels and fillers are important linguistic expressions in dialogue, but often treated as "noise" to be bypassed in modern transformer-based language models. Our work studies the representation of them in language models using three fine-tuning strategies. The models are trained on three dialogue corpora in English and Japanese, where backchannels and fillers are preserved and annotated, to investigate how fine-tuning can help LMs learn their representations. We first apply clustering analysis to the learnt representation of backchannels and fillers, and have found increased silhouette scores in representations from fine-tuned models, which suggests that fine-tuning enables LMs to distinguish the nuanced semantic variation in different backchannel and filler use. We also use natural language generation (NLG) metrics and qualitative analysis to confirm that the utterances generated by fine-tuned language models resemble human-produced utterances more closely. Our findings suggest the potentials of transforming general LMs into conversational LMs that are more capable of producing human-like languages adequately.}},
  author       = {{Wang, Yu and Lao, Leyi and Huang, Langchu and Skantze, Gabriel and Xu, Yang and Buschmeier, Hendrik}},
  booktitle    = {{Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics}},
  location     = {{San Diego, CA, USA}},
  pages        = {{5319--5348}},
  publisher    = {{Association for Computational Linguistics}},
  title        = {{{Investigating the representation of backchannels and fillers in fine-tuned language models}}},
  doi          = {{10.18653/v1/2026.acl-long.241}},
  year         = {{2026}},
}

@inproceedings{66106,
  abstract     = {{Multimodal backchannels are fundamental to conversational grounding and understanding. However, backchannels do not always transparently reflect the interlocutor's actual cognitive state. “Incongruent backchannels”, where the observable feedback implies understanding although there is no genuine understanding, are a potential source of ambiguity in any interaction. Using acoustic, head movement, and discourse-related data from 45 naturalistic dyadic interactions, we investigate whether incongruent and congruent backchannels show systematically different properties using a classification task. Results show that these backchannels are indeed separable based on multimodal features. Incongruent backchannels are typically characterised by more neutral head movement configurations and lower acoustic dynamism, while discursive cues strongly influence the classification. Overall, the findings suggest a relatively reduced effort in the signalling of incongruent backchannels.}},
  author       = {{Türk, Olcay and Lazarov, Stefan Teodorov and Wang, Yu and Grimminger, Angela and Buschmeier, Hendrik and Wagner, Petra}},
  booktitle    = {{Proceedings of INTERSPEECH 2026}},
  location     = {{Sydney, Australia}},
  title        = {{{When “yeah” means “not quite”: Multimodal detection of backchannels expressing incomplete understanding}}},
  year         = {{2026}},
}

@inbook{61150,
  abstract     = {{Since the emergence of the field of eXplainable Artificial Intelligence (XAI), a growing number of researchers have argued that XAI should consider insights from the social sciences in order to adapt explanations to the expectations and needs of human users. This has led to the emergence of a field called Social XAI, which is concerned with understanding how explanations are actively shaped in the interaction between a human user and an AI system. Recognizing this turn in XAI toward making XAI systems more “social” by providing explanations that focus on human information needs and incorporating insights from human–human explanatory interactions, in this paper we provide a formal foundation for Social XAI. We do so by proposing novel ontological accounts of the key terms used in Social XAI based on Basic Formal Ontology (BFO). Specifically, we provide novel ontological accounts for explanandum, explanans, understanding, explanation, explainer, explainee, and context. In doing so, we discuss multifaceted entities in Social XAI (having both continuant and occurrent facets; e.g., explanation) and the relationship between understanding and explanation. Additionally, we propose solutions to seemingly paradoxical views on some terms (e.g., social constructivist vs. individual constructivist perspective on explanandum).}},
  author       = {{Booshehri, Meisam and Buschmeier, Hendrik and Cimiano, Philipp}},
  booktitle    = {{Proceedings of the 15th International Conference on Formal Ontology in Information Systems}},
  isbn         = {{9781643686172}},
  issn         = {{0922-6389}},
  location     = {{Catania, Italy}},
  pages        = {{255–268}},
  publisher    = {{IOS Press}},
  title        = {{{A BFO-based ontological analysis of entities in Social XAI}}},
  doi          = {{10.3233/faia250498}},
  year         = {{2025}},
}

@inproceedings{61153,
  author       = {{Booshehri, Meisam and Buschmeier, Hendrik and Cimiano, Philipp}},
  booktitle    = {{Abstracts of the 3rd TRR 318 Conference: Contextualizing Explanations}},
  location     = {{Bielefeld, Germany}},
  title        = {{{A BFO-based ontology of context for Social XAI}}},
  year         = {{2025}},
}

@book{61178,
  editor       = {{Ilinykh, Nikolai and Robrecht, Amelie and Kopp, Stefan and Buschmeier, Hendrik}},
  issn         = {{2308-2275}},
  location     = {{Bielefeld, Germany}},
  pages        = {{271+viii}},
  title        = {{{SemDial 2025 – Bialogue. Proceedings of the 29th Workshop on the Semantics and Pragmatics of Dialogue}}},
  year         = {{2025}},
}

@misc{61429,
  author       = {{Buschmeier, Hendrik and Grimminger, Angela and Wagner, Petra and Lazarov, Stefan Teodorov and Türk, Olcay and Wang, Yu}},
  publisher    = {{LibreCat University}},
  title        = {{{MUNDEX Annotations}}},
  doi          = {{10.5281/ZENODO.17129817}},
  year         = {{2025}},
}

@phdthesis{62748,
  abstract     = {{Erklärungen spielen eine zentrale Rolle in alltäglichen persönlichen Gesprächen, indem sie den Wissensaustausch fördern, Ideen klären und das Verständnis unterstützen. In solchen Gesprächen versuchen die Erklärenden (d. h. die sachkundigere Person), das Verständnis der Explainees (d. h. die Person, die eine Erklärung erhält) durch Interaktionsprozesse wie Monitoring, Scaffolding und gemeinsame Konstruktion zu verbessern (Buschmeier et al., 2023; Rohlfing et al., 2021). Während gemeinsame Konstruktionen aus dem bidirektionalen (non-)verbalen Austausch zwischen den Gesprächspartnern entstehen, bezeichnet Scaffolding den Prozess, durch den die Erklärenden eine Erklärung anpassen, indem sie unterschiedliche Verhaltensweisen als Reaktion auf das Verhalten der Explainees einsetzen, welches deren kognitive Verarbeitung signalisiert (Wood et al., 1976). Monitoring bezeichnet einen kontinuierlichen Prozess, in dem die Gesprächspartner auf Wahrnehmungssignale wie (non-)verbale Verhaltensweisen achten, um Hinweise auf (Miss-)Verständnisse zu erkennen und zu interpretieren (Clark &amp; Krych, 2004).In der vorliegenden Arbeit berichte ich über fünf Studien zu dyadischen Erklärungen zwischen Menschen und diskutiere anhand empirischer Befunde die Interaktionsdynamiken, die bestimmten Formen verbalen und nonverbalen Erklärungsverhaltens zugrunde liegen. Zu diesem Zweck analysierte ich in den vorgestellten Studien Daten aus zwei Videokorpora zu verschiedenen Bereichen alltäglicher Erklärungen, beispielsweise medizinischen Erklärungen und Brettspielerklärungen. Das Korpus zu medizinischen Erklärungen umfasst elf naturalistische Interaktionen zwischen Ärzten und Bezugspersonen über eine bevorstehende chirurgische Operation von Kindern. Das Korpus zu Brettspielerklärungen besteht aus 87 dyadischen Brettspielerklärungen, von denen eine Teilstichprobe von 24 Interaktionen in den vorgestellten Studien spezifisch untersucht wurde.Um das verbale Erklärungsverhalten zu untersuchen, analysierte ich in zwei Studien, die sich mit medizinischen Erklärungen und Brettspielerklärungen befassten, den Zusammenhang zwischen Themenwechseln in Erklärungen und dem multimodalen Verhalten der Explainees, das von den Erklärenden beobachtet wurde. Die Analyse medizinischer Erklärungen legt nahe, dass der Wechsel von Elaborationen zu neuen Themen mit dem multimodalen Verhalten der Explainees einhergeht, welches Blickabwendung, Kopfnicken und verbale Rückkopplung umfasst. Der Wechsel zu Elaborationen hingegen ist mit einer anhaltenden Blickrichtung verbunden, unabhängig davon, ob zusätzliche Signale vorhanden sind oder nicht (Lazarov et al., 2024). Eine nachfolgende Studie zu Brettspielerklärungen (Lazarov &amp; Grimminger, in Begutachtung) erweiterte diese Analyse durch die Einbeziehung des Blickverhaltens der Erklärenden. Die Studie untersuchte den Zusammenhang zwischen gegenseitigem Blickkontakt und Blickabwendung mit der Einleitung neuer Themen. Die Ergebnisse bestätigten die Ergebnisse aus dem medizinischen Kontext: Blickabwendungen der Explainees gehen Themenwechseln häufiger voraus als gegenseitiger Blickkontakt, was mit früheren Forschungsergebnissen von Rossano (2012, 2013) übereinstimmt. Die Analyse untersuchte zudem den Zusammenhang zwischen dem / der Gesprächspartner:in, der / die Blickabwendungen initiierte, und dem / der Gesprächspartner:in, der / die Themenwechsel initiierte.Um das nonverbale Erklärungsverhalten zu untersuchen, analysierte ich die Verwendung von sprachbegleitenden Gesten von verschiedenen Erklärenden in drei Studien zu Brettspielerklärungen, in denen das zu erklärende Objekt im gemeinsamen Referenzraum physisch nicht vorhanden war. Obwohl diese Abwesenheit ein fortwährendes Bedürfnis nach der Etablierung gemeinsamer imaginärer Räume impliziert (Kang et al., 2015; Kinalzik &amp; Heller, 2021), beispielsweise durch kontinuierliches Zeigen auf unsichtbare Orte, zeigte die Studie von Lazarov &amp; Grimminger (2025), dass Gestenikonizität und zeitliche Hervorhebung auch in Themen zu Objekteigenschaften, Handlungsprozessen und bedingten Regeln variabel auftreten.Motiviert durch die kontinuierliche Verwendung von der gestischen Deixis während der physischen Abwesenheit des Explanandums erforschten die letzten zwei Studien den kognitiven Mechanismus der Anpassung des Gebrauchs deiktischer Gesten in Bezug auf die Beobachtung des Verständnisses der Explainees. Unabhängig davon, ob die Erklärenden das Verständnis der Explainees in einer retrospektiven Video Recall Aufgabe interpretierten (Lazarov &amp; Grimminger, 2024a) oder die verbalen Verständnissignale der Explainees wahrnahmen (Lazarov &amp; Grimminger, 2024b), zeigten die Analysen, dass die Häufigkeit der gestischen Deixis während der Erklärungsphase, in der das Explanandum nicht im gemeinsamen Raum vorhanden war, stabil blieb. Zu den Ergebnissen der in dieser Dissertation präsentierten Studien diskutiere ich, wie die kontinuierliche Beobachtung des Feedbackverhaltens der Explainees Anpassungen sowohl im verbalen als auch im nonverbalen Erklärungsverhalten erläutern kann. Darüber hinaus verdeutlichen die von mir präsentierten Studien das Ausmaß der individuellen Variation innerhalb und zwischen den Erklärenden, von denen jeder / jede mit drei verschiedenen Explainees interagierte.}},
  author       = {{Lazarov, Stefan Teodorov}},
  pages        = {{167}},
  publisher    = {{Universitätsbibliothek Paderborn}},
  title        = {{{The reflection of interactional monitoring in the dynamics of verbal and nonverbal forms of explaining}}},
  doi          = {{10.17619/UNIPB/1-2446}},
  year         = {{2025}},
}

@article{61156,
  abstract     = {{Explainability has become an important topic in computer science and artificial intelligence, leading to a subfield called Explainable Artificial Intelligence (XAI). The goal of providing or seeking explanations is to achieve (better) ‘understanding’ on the part of the explainee. However, what it means to ‘understand’ is still not clearly defined, and the concept itself is rarely the subject of scientific investigation. This conceptual article aims to present a model of forms of understanding for XAI-explanations and beyond. From an interdisciplinary perspective bringing together computer science, linguistics, sociology, philosophy and psychology, a definition of understanding and its forms, assessment, and dynamics during the process of giving everyday explanations are explored. Two types of understanding are considered as possible outcomes of explanations, namely enabledness, ‘knowing how’ to do or decide something, and comprehension, ‘knowing that’ – both in different degrees (from shallow to deep). Explanations regularly start with shallow understanding in a specific domain and can lead to deep comprehension and enabledness of the explanandum, which we see as a prerequisite for human users to gain agency. In this process, the increase of comprehension and enabledness are highly interdependent. Against the background of this systematization, special challenges of understanding in XAI are discussed.}},
  author       = {{Buschmeier, Hendrik and Buhl, Heike M. and Kern, Friederike and Grimminger, Angela and Beierling, Helen and Fisher, Josephine Beryl and Groß, André and Horwath, Ilona and Klowait, Nils and Lazarov, Stefan Teodorov and Lenke, Michael and Lohmer, Vivien and Rohlfing, Katharina and Scharlau, Ingrid and Singh, Amit and Terfloth, Lutz and Vollmer, Anna-Lisa and Wang, Yu and Wilmes, Annedore and Wrede, Britta}},
  journal      = {{Cognitive Systems Research}},
  keywords     = {{understanding, explaining, explanations, explainable, AI, interdisciplinarity, comprehension, enabledness, agency}},
  title        = {{{Forms of Understanding for XAI-Explanations}}},
  doi          = {{10.1016/j.cogsys.2025.101419}},
  volume       = {{94}},
  year         = {{2025}},
}

@inproceedings{61154,
  author       = {{Türk, Olcay and Lazarov, Stefan Teodorov and Buschmeier, Hendrik and Wagner, Petra and Grimminger, Angela}},
  booktitle    = {{LingCologne 2025 – Book of Abstracts}},
  location     = {{Cologne, Germany}},
  pages        = {{36}},
  title        = {{{Acoustic detection of false positive backchannels of understanding in explanations}}},
  year         = {{2025}},
}

@misc{65564,
  author       = {{Lazarov, Stefan Teodorov and Türk, Olcay and Grimminger, Angela and Wagner, Petra  and Buschmeier, Hendrik}},
  publisher    = {{LibreCat University}},
  title        = {{{Annotation Schemes Project A02 "Monitoring the understanding of explanations"}}},
  doi          = {{10.17605/OSF.IO/J2DHA}},
  year         = {{2025}},
}

@inproceedings{55917,
  abstract     = {{This work takes steps towards situating the concepts relevant to explanation and understanding in explanatory interactions within the scope of Basic Formal Ontology. We introduce novel ontological accounts of understanding and explanation in BFO-terms, which foster a shared conceptualization of explanations and explainee's understanding during explainer-explainee interactions. This approach also enables the tracking of different aspects of understanding and explanation through cognitive profiling of various measurable aspects under the heading of process profile in BFO. Additionally, we differentiate between the private mental process of understanding and understanding displays. Finally, we characterize the relationship between understanding displays and explanations.}},
  author       = {{Booshehri, Meisam and Buschmeier, Hendrik and Cimiano, Philipp}},
  booktitle    = {{Proceedings of the 4th International Workshop on Data Meets Applied Ontologies in Explainable AI (DAO-XAI)}},
  issn         = {{1613-0073}},
  location     = {{Santiago de Compostela, Spain}},
  publisher    = {{International Association for Ontology and its Applications}},
  title        = {{{Towards a BFO-based ontology of understanding in explanatory interactions}}},
  year         = {{2024}},
}

