@inproceedings{bokkahalli-satish-etal-2026-speak,
title = "Speak Your Mind: The Speech Continuation Task as a Probe of Voice-Based Model Bias",
author = "Bokkahalli Satish, Shree Harsha and
Lameris, Harm and
Perrotin, Olivier and
Henter, Gustav Eje and
Szekely, Eva",
editor = "Pranav, A and
Basile, Valerio and
Falk, Neele and
Jurgens, David and
Lapesa, Gabriella and
Lauscher, Anne and
Lo, Soda Marem",
booktitle = "Proceedings of the Second Workshop of Identity Aware {AI}",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "European Language Resources Association",
url = "https://aclanthology.org/2026.iaai-1.2/",
doi = "10.63317/3h9cs6yh6gvq",
pages = "12--19",
abstract = "Speech Continuation (SC) is the task of generating a coherent extension of a spoken prompt while preserving both semantic context and speaker identity. Because SC is constrained to a single audio stream, it offers a more direct setting for probing biases in speech foundation models than dialogue does. In this work we present the first systematic evaluation of bias in SC, investigating how gender and phonation type (breathy, creaky, end-creak) affect continuation behaviour. We evaluate three recent models: SpiritLM (base and expressive), VAE-GSLM, and SpeechGPT across speaker similarity, voice quality preservation, and text-based bias metrics. Results show that while both speaker similarity and coherence remain a challenge, textual evaluations reveal significant model and gender interactions: once coherence is sufficiently high (for VAE-GSLM), gender effects emerge on text-metrics such as agency and sentence polarity. In addition, continuations revert toward modal phonation more strongly for female prompts than for male ones, revealing a systematic voice-quality bias. These findings highlight SC as a controlled probe of socially relevant representational biases in speech foundation models, and suggest that it will become an increasingly informative diagnostic as continuation quality improves."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="bokkahalli-satish-etal-2026-speak">
<titleInfo>
<title>Speak Your Mind: The Speech Continuation Task as a Probe of Voice-Based Model Bias</title>
</titleInfo>
<name type="personal">
<namePart type="given">Shree</namePart>
<namePart type="given">Harsha</namePart>
<namePart type="family">Bokkahalli Satish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Harm</namePart>
<namePart type="family">Lameris</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Olivier</namePart>
<namePart type="family">Perrotin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gustav</namePart>
<namePart type="given">Eje</namePart>
<namePart type="family">Henter</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Szekely</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second Workshop of Identity Aware AI</title>
</titleInfo>
<name type="personal">
<namePart type="given">A</namePart>
<namePart type="family">Pranav</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Valerio</namePart>
<namePart type="family">Basile</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Neele</namePart>
<namePart type="family">Falk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Jurgens</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gabriella</namePart>
<namePart type="family">Lapesa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anne</namePart>
<namePart type="family">Lauscher</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Soda</namePart>
<namePart type="given">Marem</namePart>
<namePart type="family">Lo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Language Resources Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Speech Continuation (SC) is the task of generating a coherent extension of a spoken prompt while preserving both semantic context and speaker identity. Because SC is constrained to a single audio stream, it offers a more direct setting for probing biases in speech foundation models than dialogue does. In this work we present the first systematic evaluation of bias in SC, investigating how gender and phonation type (breathy, creaky, end-creak) affect continuation behaviour. We evaluate three recent models: SpiritLM (base and expressive), VAE-GSLM, and SpeechGPT across speaker similarity, voice quality preservation, and text-based bias metrics. Results show that while both speaker similarity and coherence remain a challenge, textual evaluations reveal significant model and gender interactions: once coherence is sufficiently high (for VAE-GSLM), gender effects emerge on text-metrics such as agency and sentence polarity. In addition, continuations revert toward modal phonation more strongly for female prompts than for male ones, revealing a systematic voice-quality bias. These findings highlight SC as a controlled probe of socially relevant representational biases in speech foundation models, and suggest that it will become an increasingly informative diagnostic as continuation quality improves.</abstract>
<identifier type="citekey">bokkahalli-satish-etal-2026-speak</identifier>
<identifier type="doi">10.63317/3h9cs6yh6gvq</identifier>
<location>
<url>https://aclanthology.org/2026.iaai-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>12</start>
<end>19</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Speak Your Mind: The Speech Continuation Task as a Probe of Voice-Based Model Bias
%A Bokkahalli Satish, Shree Harsha
%A Lameris, Harm
%A Perrotin, Olivier
%A Henter, Gustav Eje
%A Szekely, Eva
%Y Pranav, A.
%Y Basile, Valerio
%Y Falk, Neele
%Y Jurgens, David
%Y Lapesa, Gabriella
%Y Lauscher, Anne
%Y Lo, Soda Marem
%S Proceedings of the Second Workshop of Identity Aware AI
%D 2026
%8 May
%I European Language Resources Association
%C Palma de Mallorca, Spain
%F bokkahalli-satish-etal-2026-speak
%X Speech Continuation (SC) is the task of generating a coherent extension of a spoken prompt while preserving both semantic context and speaker identity. Because SC is constrained to a single audio stream, it offers a more direct setting for probing biases in speech foundation models than dialogue does. In this work we present the first systematic evaluation of bias in SC, investigating how gender and phonation type (breathy, creaky, end-creak) affect continuation behaviour. We evaluate three recent models: SpiritLM (base and expressive), VAE-GSLM, and SpeechGPT across speaker similarity, voice quality preservation, and text-based bias metrics. Results show that while both speaker similarity and coherence remain a challenge, textual evaluations reveal significant model and gender interactions: once coherence is sufficiently high (for VAE-GSLM), gender effects emerge on text-metrics such as agency and sentence polarity. In addition, continuations revert toward modal phonation more strongly for female prompts than for male ones, revealing a systematic voice-quality bias. These findings highlight SC as a controlled probe of socially relevant representational biases in speech foundation models, and suggest that it will become an increasingly informative diagnostic as continuation quality improves.
%R 10.63317/3h9cs6yh6gvq
%U https://aclanthology.org/2026.iaai-1.2/
%U https://doi.org/10.63317/3h9cs6yh6gvq
%P 12-19
Markdown (Informal)
[Speak Your Mind: The Speech Continuation Task as a Probe of Voice-Based Model Bias](https://aclanthology.org/2026.iaai-1.2/) (Bokkahalli Satish et al., iaai 2026)
ACL