Contrastive explanations, which indicate why an AI system produced one output (the target) instead of another (the foil), are widely regarded in explainable AI as more informative and interpretable than standard explanations. However, obtaining such explanations for speech-to-text (S2T) generative models remains an open challenge. Drawing from feature attribution techniques, we propose the first method to obtain contrastive explanations in S2T by analyzing how parts of the input spectrogram influence the choice between alternative outputs. Through a case study on gender assignment in speech translation, we show that our method accurately identifies the audio features that drive the selection of one gender over another. By extending the scope of contrastive explanations to S2T, our work provides a foundation for better understanding S2T models.
@article{arxiv.2509.26543,
title = {The Unheard Alternative: Contrastive Explanations for Speech-to-Text Models},
author = {Lina Conti and Dennis Fucci and Marco Gaido and Matteo Negri and Guillaume Wisniewski and Luisa Bentivogli},
journal= {arXiv preprint arXiv:2509.26543},
year = {2026}
}