@inproceedings{29290,
  abstract     = {{Classifying nodes in knowledge graphs is an important task, e.g., predicting
missing types of entities, predicting which molecules cause cancer, or
predicting which drugs are promising treatment candidates. While black-box
models often achieve high predictive performance, they are only post-hoc and
locally explainable and do not allow the learned model to be easily enriched
with domain knowledge. Towards this end, learning description logic concepts
from positive and negative examples has been proposed. However, learning such
concepts often takes a long time and state-of-the-art approaches provide
limited support for literal data values, although they are crucial for many
applications. In this paper, we propose EvoLearner - an evolutionary approach
to learn ALCQ(D), which is the attributive language with complement (ALC)
paired with qualified cardinality restrictions (Q) and data properties (D). We
contribute a novel initialization method for the initial population: starting
from positive examples (nodes in the knowledge graph), we perform biased random
walks and translate them to description logic concepts. Moreover, we improve
support for data properties by maximizing information gain when deciding where
to split the data. We show that our approach significantly outperforms the
state of the art on the benchmarking framework SML-Bench for structured machine
learning. Our ablation study confirms that this is due to our novel
initialization method and support for data properties.}},
  author       = {{Heindorf, Stefan and Blübaum, Lukas and Düsterhus, Nick and Werner, Till and Golani, Varun Nandkumar and Demir, Caglar and Ngonga Ngomo, Axel-Cyrille}},
  booktitle    = {{WWW}},
  pages        = {{818--828}},
  publisher    = {{ACM}},
  title        = {{{EvoLearner: Learning Description Logics with Evolutionary Algorithms}}},
  doi          = {{10.1145/3485447.3511925}},
  year         = {{2022}},
}

@article{29851,
  author       = {{Pestryakova, Svetlana  and Vollmers, Daniel and Sherif, Mohamed and Heindorf, Stefan and Saleem, Muhammad  and Moussallem, Diego and Ngonga Ngomo, Axel-Cyrille}},
  journal      = {{Scientific Data}},
  title        = {{{CovidPubGraph: A FAIR Knowledge Graph of COVID-19 Publications}}},
  doi          = {{10.1038/s41597-022-01298-2}},
  year         = {{2022}},
}

@inproceedings{34674,
  abstract     = {{Smart home systems contain plenty of features that enhance wellbeing in everyday life through artificial intelligence (AI). However, many users feel insecure because they do not understand the AI’s functionality and do not feel they are in control of it. Combining technical, psychological and philosophical views on AI, we rethink smart homes as interactive systems where users can partake in an intelligent agent’s learning. Parallel to the goals of explainable AI (XAI), we explored the possibility of user involvement in supervised learning of the smart home to have a first approach to improve acceptance, support subjective understanding and increase perceived control. In this work, we conducted two studies: In an online pre-study, we asked participants about their attitude towards teaching AI via a questionnaire. In the main study, we performed a Wizard of Oz laboratory experiment with human participants, where participants spent time in a prototypical smart home and taught activity recognition to the intelligent agent through supervised learning based on the user’s behaviour. We found that involvement in the AI’s learning phase enhanced the users’ feeling of control, perceived understanding and perceived usefulness of AI in general. The participants reported positive attitudes towards training a smart home AI and found the process understandable and controllable. We suggest that involving the user in the learning phase could lead to better personalisation and increased understanding and control by users of intelligent agents for smart home automation.}},
  author       = {{Sieger, Leonie Nora and Hermann, Julia and Schomäcker, Astrid and Heindorf, Stefan and Meske, Christian and Hey, Celine-Chiara and Doğangün, Ayşegül}},
  booktitle    = {{International Conference on Human-Agent Interaction}},
  keywords     = {{human-agent interaction, smart homes, supervised learning, participation}},
  location     = {{Christchurch, New Zealand}},
  publisher    = {{ACM}},
  title        = {{{User Involvement in Training Smart Home Agents}}},
  doi          = {{10.1145/3527188.3561914}},
  year         = {{2022}},
}

@inbook{33738,
  author       = {{Zahera, Hamada Mohamed Abdelsamee and Heindorf, Stefan and Balke, Stefan and Haupt, Jonas and Voigt, Martin and Walter, Carolin and Witter, Fabian and Ngonga Ngomo, Axel-Cyrille}},
  booktitle    = {{The Semantic Web: ESWC 2022 Satellite Events}},
  isbn         = {{9783031116087}},
  issn         = {{0302-9743}},
  publisher    = {{Springer International Publishing}},
  title        = {{{Tab2Onto: Unsupervised Semantification with Knowledge Graph Embeddings}}},
  doi          = {{10.1007/978-3-031-11609-4_9}},
  year         = {{2022}},
}

@inproceedings{33739,
  abstract     = {{At least 5% of questions submitted to search engines ask about cause-effect relationships in some way. To support the development of tailored approaches that can answer such questions, we construct Webis-CausalQA-22, a benchmark corpus of 1.1 million causal questions with answers. We distinguish different types of causal questions using a novel typology derived from a data-driven, manual analysis of questions from ten large question answering (QA) datasets. Using high-precision lexical rules, we extract causal questions of each type from these datasets to create our corpus. As an initial baseline, the state-of-the-art QA model UnifiedQA achieves a ROUGE-L F1 score of 0.48 on our new benchmark.}},
  author       = {{Bondarenko, Alexander and Wolska, Magdalena and Heindorf, Stefan and Blübaum, Lukas and Ngonga Ngomo, Axel-Cyrille and Stein, Benno and Braslavski, Pavel and Hagen, Matthias and Potthast, Martin}},
  booktitle    = {{Proceedings of the 29th International Conference on Computational Linguistics}},
  pages        = {{3296–3308}},
  publisher    = {{International Committee on Computational Linguistics}},
  title        = {{{CausalQA: A Benchmark for Causal Question Answering}}},
  year         = {{2022}},
}

@inbook{29292,
  author       = {{Feldhans, Robert and Wilke, Adrian and Heindorf, Stefan and Shaker, Mohammad Hossein and Hammer, Barbara and Ngonga Ngomo, Axel-Cyrille and Hüllermeier, Eyke}},
  booktitle    = {{Intelligent Data Engineering and Automated Learning – IDEAL 2021}},
  isbn         = {{9783030916077}},
  issn         = {{0302-9743}},
  publisher    = {{Springer International Publishing}},
  title        = {{{Drift Detection in Text Data with Document Embeddings}}},
  doi          = {{10.1007/978-3-030-91608-4_11}},
  year         = {{2021}},
}

@inproceedings{29287,
  abstract     = {{Knowledge graph embedding research has mainly focused on the two smallest
normed division algebras, $\mathbb{R}$ and $\mathbb{C}$. Recent results suggest
that trilinear products of quaternion-valued embeddings can be a more effective
means to tackle link prediction. In addition, models based on convolutions on
real-valued embeddings often yield state-of-the-art results for link
prediction. In this paper, we investigate a composition of convolution
operations with hypercomplex multiplications. We propose the four approaches
QMult, OMult, ConvQ and ConvO to tackle the link prediction problem. QMult and
OMult can be considered as quaternion and octonion extensions of previous
state-of-the-art approaches, including DistMult and ComplEx. ConvQ and ConvO
build upon QMult and OMult by including convolution operations in a way
inspired by the residual learning framework. We evaluated our approaches on
seven link prediction datasets including WN18RR, FB15K-237 and YAGO3-10.
Experimental results suggest that the benefits of learning hypercomplex-valued
vector representations become more apparent as the size and complexity of the
knowledge graph grows. ConvO outperforms state-of-the-art approaches on
FB15K-237 in MRR, Hit@1 and Hit@3, while QMult, OMult, ConvQ and ConvO
outperform state-of-the-approaches on YAGO3-10 in all metrics. Results also
suggest that link prediction performances can be further improved via
prediction averaging. To foster reproducible research, we provide an
open-source implementation of approaches, including training and evaluation
scripts as well as pretrained models.}},
  author       = {{Demir, Caglar and Moussallem, Diego and Heindorf, Stefan and Ngonga Ngomo, Axel-Cyrille}},
  booktitle    = {{The 13th Asian Conference on Machine Learning, ACML 2021}},
  title        = {{{Convolutional Hypercomplex Embeddings for Link Prediction}}},
  year         = {{2021}},
}

@inproceedings{29294,
  author       = {{Nickchen, Tobias and Heindorf, Stefan and Engels, Gregor}},
  booktitle    = {{2021 IEEE Winter Conference on Applications of Computer Vision (WACV)}},
  publisher    = {{IEEE}},
  title        = {{{Generating Physically Sound Training Data for Image Recognition of Additively Manufactured Parts}}},
  doi          = {{10.1109/wacv48630.2021.00204}},
  year         = {{2021}},
}

@misc{33733,
  author       = {{Heindorf, Stefan}},
  title        = {{{Automatically generating instructions from tutorials for search and user navigation}}},
  year         = {{2021}},
}

@inproceedings{29291,
  author       = {{Zahera, Hamada Mohamed Abdelsamee and Heindorf, Stefan and Ngonga Ngomo, Axel-Cyrille}},
  booktitle    = {{Proceedings of the 11th on Knowledge Capture Conference}},
  publisher    = {{ACM}},
  title        = {{{ASSET: A Semi-supervised Approach for Entity Typing in Knowledge Graphs}}},
  doi          = {{10.1145/3460210.3493563}},
  year         = {{2021}},
}

@inproceedings{20141,
  author       = {{Heindorf, Stefan and Scholten, Yan and Wachsmuth, Henning and Ngonga Ngomo, Axel-Cyrille and Potthast, Martin}},
  booktitle    = {{Proceedings of the 28th ACM International Conference on Information and Knowledge Management (CIKM 2020)}},
  pages        = {{3023--3030}},
  title        = {{{CauseNet: Towards a Causality Graph Extracted from the Web}}},
  doi          = {{10.1145/3340531.3412763}},
  year         = {{2020}},
}

@inproceedings{7668,
  author       = {{Heindorf, Stefan and Scholten, Yan and Engels, Gregor and Potthast, Martin}},
  booktitle    = {{WWW}},
  location     = {{San Francisco, USA}},
  pages        = {{670--680}},
  publisher    = {{ACM}},
  title        = {{{Debiasing Vandalism Detection Models at Wikidata}}},
  doi          = {{10.1145/3308558.3313507}},
  year         = {{2019}},
}

@phdthesis{15333,
  author       = {{Heindorf, Stefan}},
  publisher    = {{Universität Paderborn}},
  title        = {{{Vandalism Detection in Crowdsourced Knowledge Bases}}},
  year         = {{2019}},
}

@inproceedings{14568,
  author       = {{Heindorf, Stefan and Scholten, Yan and Engels, Gregor and Potthast, Martin}},
  booktitle    = {{INFORMATIK}},
  pages        = {{289--290}},
  title        = {{{Debiasing Vandalism Detection Models at Wikidata (Extended Abstract)}}},
  doi          = {{10.18420/inf2019_48}},
  year         = {{2019}},
}

@inproceedings{5831,
  abstract     = {{Many websites offer links to social media sites for convenient content sharing. Unfortunately, those sharing capabilities are quite restricted and it is seldom possible to share content with other services, like those provided by a user's favorite applications or smart devices. In this paper, we present Semantic Data Mediator (SDM) --- a flexible middleware linking a vast number of services to millions of websites. Based on reusable repositories of service descriptions defined by the crowd, users can easily fill a personal registry with their favorite services, which can then be linked to websites by SDM. For this, SDM leverages semantic data, which is already available on millions of websites due to search engine optimization. Further support for our approach from website or service developers is not required. To enable the use of a broad range of services, data conversion services are automatically composed by SDM to transform data according to the needs of the different services. In addition to linking web services, various service adapters allow services of applications and smart devices to be linked as well. We have fully implemented our approach and present a real-world case study demonstrating its feasibility and usefulness.}},
  author       = {{Wolters, Dennis and Heindorf, Stefan and Kirchhoff, Jonas and Engels, Gregor}},
  booktitle    = {{Service-Oriented Computing -- ICSOC 2017 Workshops}},
  editor       = {{Braubach, Lars and Murillo, Juan M. and Kaviani, Nima and Lama, Manuel and Burgueño, Loli and Moha, Naouel and Oriol, Marc}},
  isbn         = {{978-3-319-91764-1}},
  pages        = {{388--392}},
  publisher    = {{Springer International Publishing}},
  title        = {{{Semantic Data Mediator: Linking Services to Websites}}},
  doi          = {{10.1007/978-3-319-91764-1_36}},
  year         = {{2018}},
}

@inproceedings{5829,
  abstract     = {{Websites increasingly embed semantic data for search engine optimization. The most common ontology for semantic data, schema.org, is supported by all major search engines and describes over 500 data types, including calendar events, recipes, products, and TV shows. As of today, users wishing to pass this data to their favorite applications, e.g., their calendars, cookbooks, price comparison applications or even smart devices such as TV receivers, rely on cumbersome and error-prone workarounds such as reentering the data or a series of copy and paste operations. In this paper, we present Semantic Data Mediator (SDM), an approach that allows the easy transfer of semantic data to a multitude of services, ranging from web services to applications installed on different devices. SDM extracts semantic data from the currently displayed web page on the client-side, offers suitable services to the user, and by the press of a button, forwards this data to the desired service while doing all the necessary data conversion and service interface adaptation in between. To realize this, we built a reusable repository of service descriptions, data converters, and service adapters, which can be extended by the crowd. Our approach for linking services to websites relies solely on semantic data and does not require any additional support by either website or service developers. We have fully implemented our approach and present a real-world case study demonstrating its feasibility and usefulness.}},
  author       = {{Wolters, Dennis and Heindorf, Stefan and Kirchhoff, Jonas and Engels, Gregor}},
  booktitle    = {{2017 IEEE International Conference on Web Services (ICWS)}},
  editor       = {{Altintas, Ilkay and Chen, Shiping}},
  isbn         = {{9781538607527}},
  keywords     = {{Services, Websites, Semantic Data, schema.org, Data Conversion, Interface Adaptation, Mediation}},
  publisher    = {{IEEE}},
  title        = {{{Linking Services to Websites by Leveraging Semantic Data}}},
  doi          = {{10.1109/icws.2017.80}},
  year         = {{2017}},
}

@inproceedings{6722,
  abstract     = {{We report on the Wikidata vandalism detection task at the WSDM Cup 2017. The
task received five submissions for which this paper describes their evaluation
and a comparison to state of the art baselines. Unlike previous work, we recast
Wikidata vandalism detection as an online learning problem, requiring
participant software to predict vandalism in near real-time. The
best-performing approach achieves a ROC-AUC of 0.947 at a PR-AUC of 0.458. In
particular, this task was organized as a software submission task: to maximize
reproducibility as well as to foster future research and development on this
task, the participants were asked to submit their working software to the TIRA
experimentation platform along with the source code for open source release.}},
  author       = {{Heindorf, Stefan and Potthast, Martin and Engels, Gregor and Stein, Benno}},
  booktitle    = {{WSDM Cup 2017 Notebook Papers}},
  title        = {{{Overview of the Wikidata Vandalism Detection Task at WSDM Cup 2017}}},
  year         = {{2017}},
}

@unpublished{33732,
  abstract     = {{The WSDM Cup 2017 was a data mining challenge held in conjunction with the
10th International Conference on Web Search and Data Mining (WSDM). It
addressed key challenges of knowledge bases today: quality assurance and entity
search. For quality assurance, we tackle the task of vandalism detection, based
on a dataset of more than 82 million user-contributed revisions of the Wikidata
knowledge base, all of which annotated with regard to whether or not they are
vandalism. For entity search, we tackle the task of triple scoring, using a
dataset that comprises relevance scores for triples from type-like relations
including occupation and country of citizenship, based on about 10,000 human
relevance judgements. For reproducibility sake, participants were asked to
submit their software on TIRA, a cloud-based evaluation platform, and they were
incentivized to share their approaches open source.}},
  author       = {{Potthast, Martin and Heindorf, Stefan and Bast, Hannah}},
  booktitle    = {{arXiv:1712.09528}},
  title        = {{{Proceedings of the WSDM Cup 2017: Vandalism Detection and Triple Scoring}}},
  year         = {{2017}},
}

@inproceedings{6721,
  author       = {{Heindorf, Stefan and Potthast, Martin and Bast, Hannah and Buchhold, Björn and Haussmann, Elmar}},
  booktitle    = {{WSDM}},
  pages        = {{827--828}},
  publisher    = {{ACM}},
  title        = {{{WSDM Cup 2017: Vandalism Detection and Triple Scoring}}},
  doi          = {{10.1145/3018661.3022762}},
  year         = {{2017}},
}

@inproceedings{137,
  abstract     = {{Wikidata is the new, large-scale knowledge base of the Wikimedia Foundation. Its knowledge is increasingly used within Wikipedia itself and various other kinds of information systems, imposing high demands on its integrity.Wikidata can be edited by anyone and, unfortunately, it frequently gets vandalized, exposing all information systems using it to the risk of spreading vandalized and falsified information. In this paper, we present a new machine learning-based approach to detect vandalism in Wikidata.We propose a set of 47 features that exploit both content and context information, and we report on 4 classifiers of increasing effectiveness tailored to this learning task. Our approach is evaluated on the recently published Wikidata Vandalism Corpus WDVC-2015 and it achieves an area under curve value of the receiver operating characteristic, ROC-AUC, of 0.991. It significantly outperforms the state of the art represented by the rule-based Wikidata Abuse Filter (0.865 ROC-AUC) and a prototypical vandalism detector recently introduced by Wikimedia within the Objective Revision Evaluation Service (0.859 ROC-AUC).}},
  author       = {{Heindorf, Stefan and Potthast, Matthias and Stein, Benno and Engels, Gregor}},
  booktitle    = {{Proceedings of the 25th International Conference on Information and Knowledge Management (CIKM 2016)}},
  pages        = {{327----336}},
  title        = {{{Vandalism Detection in Wikidata}}},
  doi          = {{10.1145/2983323.2983740}},
  year         = {{2016}},
}

