@inproceedings{Gugliotta-Dinarelli:TALN:2020,
    author = "Gugliotta, Elisa and Dinarelli, Marco",
    title = "TArC. Un corpus d'arabish tunisien",
    booktitle = "Actes de la Conf\'erence sur le Traitement Automatique des Langues Naturelles. Volume 2 : Traitement Automatique des Langues Naturelles (Articles courts)",
    month = "6",
    year = "2020",
    address = "Nancy, France",
    publisher = "Association pour le Traitement Automatique des Langues",
    pages = "232-240",
    note = "Cet article d\'ecrit la proc\'edure de constitution du premier corpus d'arabish tunisien (TArC) annot\'e avec des informations morpho-syntaxiques",
    abstract = "TArC : Incrementally and Semi-Automatically Collecting a Tunisian arabish Corpus This article describes the collection process of the first morpho-syntactically annotated Tunisian arabish Corpus (TArC). Arabish is a spontaneous coding of Arabic Dialects (AD) in Latin characters and arithmographs (numbers used as letters). This code-system was developed by Arabic-speaking users of social media in order to facilitate the communication on digital devices. Arabish differs for each Arabic dialect and each arabish code-system is under-resourced. In the last few years, the attention of NLP on AD has considerably increased. TArC will be thus a useful support for different types of analyses, as well as for NLP tools training. In this article we will describe preliminary work on the TArC semi-automatic construction process and some of the first analyses on the corpus. In order to provide a complete overview of the challenges faced during the building process, we will present the main Tunisian dialect characteristics and its encoding in Tunisian arabish.",
    keywords = "Tunisian arabish Corpus, Arabic Dialect, Arabizi.  Volume 2 : Traitement Automatique des Langues Naturelles, pages 232-240.",
    url = "http://talnarchives.atala.org/TALN/TALN-2020/133.pdf"
}
