@inproceedings{84fd1a0377ad480da65d872385295e56,
title = "Exploiting Dialect Identification in Automatic Dialectal Text Normalization",
abstract = "Dialectal Arabic is the primary spoken language used by native Arabic speakers in daily communication. The rise of social media platforms has notably expanded its use as a written language. However, Arabic dialects do not have standard orthographies. This, combined with the inherent noise in user-generated content on social media, presents a major challenge to NLP applications dealing with Dialectal Arabic. In this paper, we explore and report on the task of CODAfication, which aims to normalize Dialectal Arabic into the Conventional Orthography for Dialectal Arabic (CODA). We work with a unique parallel corpus of multiple Arabic dialects focusing on five major city dialects. We benchmark newly developed pretrained sequence-to-sequence models on the task of CODAfication. We further show that using dialect identification information improves the performance across all dialects. We make our code, data, and pretrained models publicly available.",
author = "Bashar Alhafni and Sarah Al-Towaity and Ziyad Fawzy and Fatema Nassar and Fadhl Eryani and Houda Bouamor and Nizar Habash",
note = "Publisher Copyright: {\textcopyright}2024 Association for Computational Linguistics.; 2nd Arabic Natural Language Processing Conference, ArabicNLP 2024 ; Conference date: 16-08-2024",
year = "2024",
language = "English (US)",
series = "ArabicNLP 2024 - 2nd Arabic Natural Language Processing Conference, Proceedings of the Conference",
publisher = "Association for Computational Linguistics (ACL)",
pages = "42--54",
editor = "Nizar Habash and Houda Bouamor and Ramy Eskander and Nadi Tomeh and Farha, {Ibrahim Abu} and Ahmed Abdelali and Samia Touileb and Injy Hamed and Yaser Onaizan and Bashar Alhafni and Wissam Antoun and Salam Khalifa and Hatem Haddad and Imed Zitouni and Badr AlKhamissi and Rawan Almatham and Khalil Mrini",
booktitle = "ArabicNLP 2024 - 2nd Arabic Natural Language Processing Conference, Proceedings of the Conference",
}