@inproceedings{d39bfeb980c24b3b82098af4daac3e6c,
title = "Quantifying the Bias of Transformer-Based Language Models for African American English in Masked Language Modeling",
abstract = "In recent years, groundbreaking transformer-based language models (LMs) have made tremendous advances in natural language processing (NLP) tasks. However, the measurement of their fairness with respect to different social groups still remains unsolved. In this paper, we propose and thoroughly validate an evaluation technique to assess the quality and bias of language model predictions on transcripts of both spoken African American English (AAE) and Spoken American English (SAE). Our analysis reveals the presence of a bias towards SAE encoded by state-of-the-art LMs such as BERT and DistilBERT and a lower bias in distilled LMs. We also observe a bias towards AAE in RoBERTa and BART. Additionally, we show evidence that this disparity is present across all the LMs when we only consider the grammar and the syntax specific to AAE.",
keywords = "Bias and Fairness, Evaluation, Language Model, Transformers",
author = "Flavia Salutari and Jerome Ramos and Rahmani, \{Hossein A.\} and Leonardo Linguaglossa and Aldo Lipani",
note = "Publisher Copyright: {\textcopyright} 2023, The Author(s), under exclusive license to Springer Nature Switzerland AG.; 27th Pacific-Asia Conference on Knowledge Discovery and Data Mining, PAKDD 2023 ; Conference date: 25-05-2023 Through 28-05-2023",
year = "2023",
month = jan,
day = "1",
doi = "10.1007/978-3-031-33374-3\_42",
language = "English",
isbn = "9783031333736",
series = "Lecture Notes in Computer Science",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "532--543",
editor = "Hisashi Kashima and Tsuyoshi Ide and Wen-Chih Peng",
booktitle = "Advances in Knowledge Discovery and Data Mining - 27th Pacific-Asia Conference on Knowledge Discovery and Data Mining, PAKDD 2023, Proceedings",
}