@inproceedings{LuengenKupietz2017, author = {Harald L{\"u}ngen and Marc Kupietz}, title = {CMC Corpora in DeReKo}, series = {Proceedings of the Workshop on Challenges in the Management of Large Corpora and Big Data and Natural Language Processing (CMLC-5+BigNLP) 2017 including the papers from the Web-as-Corpus (WAC-XI) guest section. Birmingham, 24 July 2017}, editor = {Piotr BaƄski and Marc Kupietz and Harald L{\"u}ngen and Paul Rayson and Hanno Biber and Evelyn Breiteneder and Simon Clematide and John Mariani and Mark Stevenson and Theresa Sick}, publisher = {Institut f{\"u}r Deutsche Sprache}, address = {Mannheim}, url = {http://nbn-resolving.de/urn:nbn:de:bsz:mh39-62592}, pages = {20 -- 24}, year = {2017}, abstract = {We introduce three types of corpora of computer-mediated communication that have recently been compiled at the Institute for the German Language or curated from an external project and included in DeReKo, the German Reference Corpus, namely Wikipedia (discussion) corpora, the Usenet news corpus, and the Dortmund Chat Corpus. The data and corpora have been converted to I5, the TEI customization to represent texts in DeReKo, and are researchable via the web-based IDS corpus research interfaces and in the case of Wikipedia and chat also downloadable from the IDS repository and download server, respectively.}, language = {en} }