,
Alberto Abad
,
Helena Moniz
Creative Commons Attribution 4.0 International license
Recent advances in text-to-speech (TTS) have led to synthetic speech that is often indistinguishable from natural speech at the level of individual utterances. However, it remains unclear whether such systems reproduce the prosodic variability observed in natural speech in a large native-speaker population. This question is particularly relevant for applications in computer-assisted language learning (CALL), as variability is a central property of prosody and a prerequisite for robust assessment. In this work, we investigate whether TTS can approximate the distribution of prosodic patterns found in native speech, and whether it can serve as a reference for evaluating second language (L2) productions. To this end, we first compare different TTS models to native speakers, and then compare L2 speakers to synthetic speech by the model previously seen as the closest to native speakers. Our results show that regardless of TTS achieving high perceptual quality, its prosodic variability substantially differs from that of native speakers. As a consequence, comparisons between L2 speech and TTS-based references reveal both the potential and the limitations of using synthetic speech for prosody assessment.
@InProceedings{juliao_et_al:OASIcs.SLATE.2026.2,
author = {Juli\~{a}o, Mariana and Abad, Alberto and Moniz, Helena},
title = {{Evaluating the Prosodic Diversity of TTS Models for L2 Prosody Assessment}},
booktitle = {15th Symposium on Languages, Applications and Technologies (SLATE 2026)},
pages = {2:1--2:13},
series = {Open Access Series in Informatics (OASIcs)},
ISBN = {978-3-95977-440-6},
ISSN = {2190-6807},
year = {2026},
volume = {144},
editor = {Batista, Fernando and Ribeiro, Eug\'{e}nio and Ribeiro, Ricardo and Santos, Andr\'{e} L.},
publisher = {Schloss Dagstuhl -- Leibniz-Zentrum f{\"u}r Informatik},
address = {Dagstuhl, Germany},
URL = {https://drops.dagstuhl.de/entities/document/10.4230/OASIcs.SLATE.2026.2},
URN = {urn:nbn:de:0030-drops-267007},
doi = {10.4230/OASIcs.SLATE.2026.2},
annote = {Keywords: Speech synthesis, TTS, prosody, L1, L2}
}