@inproceedings{10284,
  abstract     = {{We study text reuse related to Wikipedia at scale by compiling the first corpus of text reuse cases within Wikipedia as well as without (i.e., reuse of Wikipedia text in a sample of the Common Crawl). To discover reuse beyond verbatim copy and paste, we employ state-of-the-art text reuse detection technology, scaling it for the first time to process the entire Wikipedia as part of a distributed retrieval pipeline. We further report on a pilot analysis of the 100 million reuse cases inside, and the 1.6 million reuse cases outside Wikipedia that we discovered. Text reuse inside Wikipedia gives rise to new tasks such as article template induction, fixing quality flaws, or complementing Wikipedia's ontology. Text reuse outside Wikipedia yields a tangible metric for the emerging field of quantifying Wikipedia's influence on the web. To foster future research into these tasks, and for reproducibility's sake, the Wikipedia text reuse corpus and the retrieval pipeline are made freely available.}},
  author       = {{Alshomary, Milad and Völske, Michael and Licht, Tristan and Wachsmuth, Henning and Stein, Benno and Hagen, Matthias and Potthast, Martin}},
  booktitle    = {{Advances in Information Retrieval}},
  editor       = {{Azzopardi, Leif and Stein, Benno and Fuhr, Norbert and Mayr, Philipp and Hauff, Claudia and Hiemstra, Djoerd}},
  isbn         = {{978-3-030-15712-8}},
  pages        = {{747--754}},
  publisher    = {{Springer International Publishing}},
  title        = {{{Wikipedia Text Reuse: Within and Without}}},
  year         = {{2019}},
}

@article{10331,
  author       = {{Kiesel, Johannes and Kneist, Florian and Alshomary, Milad and Stein, Benno and Hagen, Matthias and Potthast, Martin}},
  issn         = {{1936-1955}},
  journal      = {{Journal of Data and Information Quality}},
  pages        = {{1--25}},
  title        = {{{Reproducible Web Corpora}}},
  doi          = {{10.1145/3239574}},
  year         = {{2018}},
}

@inproceedings{3904,
  author       = {{Hagen, Matthias and Kiesel, Johannes and Alshomary, Milad and Stein, Benno}},
  booktitle    = {{Working Notes of CLEF 2017 - Conference and Labs of the Evaluation Forum}},
  title        = {{{Webis at the CLEF 2017 Dynamic Search Lab}}},
  year         = {{2017}},
}

@article{3905,
  author       = {{Abu Quba Rana, Chamsi and Hassas, Salima and Usama, Fayyad and Alshomary, Milad and Gertosio, Christine}},
  journal      = {{2014 IEEE/ACS 11th International Conference on Computer Systems and Applications (AICCSA)}},
  pages        = {{169--175}},
  title        = {{{iSoNTRE: The Social Network Transformer into Recommendation Engine}}},
  year         = {{2014}},
}

