@inproceedings{albayrak-2026-winotr,
title = "{W}ino{TR}: Evaluating Gender Bias in Machine Translation from a Gender-Neutral Language Using Causal Inference",
author = "Albayrak, Deniz",
editor = "Lardelli, Manuel and
Savoldi, Beatrice and
Hackenbuchner, Jani{\c{c}}a and
Bentivogli, Luisa and
Gkovedarou, Eleni and
Daems, Joke",
booktitle = "Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies ({GITT} 2026)",
month = jun,
year = "2026",
address = "Tilburg, the Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.gitt-1.9/",
pages = "100--107",
abstract = "We present WinoTR, a Turkish adaptation of the WinoMT challenge dataset (Stanovsky et al., 2019). While WinoMT has been widely studied across multiple languages, its adaptation to Turkish {---} a morphologically rich language with no grammatical gender {---} and its analysis through a causal lens remain unexplored. Using 4,752 sentences adapted from the original dataset across pro-stereotypical, anti-stereotypical, and neutral conditions, we apply Double Machine Learning (DML) to estimate the causal effect of gender cues on stereotype-consistent translation output. Our results reveal a striking asymmetry: cue direction has a large and statistically significant effect on translation outcomes, while cue presence alone produces virtually no effect. Even without any gender signal, MT systems default to stereotype-consistent translations in 62.9{\%} of cases across three systems (DeepL, Google Translate, OpenAI). By leveraging this typological property, our causal analysis reveals that gender bias in contemporary MT and LLM-based translation systems runs deeper than surface-level cue processing, persisting as an embedded prior independent of any explicit gender signal in the input."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="albayrak-2026-winotr">
<titleInfo>
<title>WinoTR: Evaluating Gender Bias in Machine Translation from a Gender-Neutral Language Using Causal Inference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Deniz</namePart>
<namePart type="family">Albayrak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Beatrice</namePart>
<namePart type="family">Savoldi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luisa</namePart>
<namePart type="family">Bentivogli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eleni</namePart>
<namePart type="family">Gkovedarou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joke</namePart>
<namePart type="family">Daems</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, the Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present WinoTR, a Turkish adaptation of the WinoMT challenge dataset (Stanovsky et al., 2019). While WinoMT has been widely studied across multiple languages, its adaptation to Turkish — a morphologically rich language with no grammatical gender — and its analysis through a causal lens remain unexplored. Using 4,752 sentences adapted from the original dataset across pro-stereotypical, anti-stereotypical, and neutral conditions, we apply Double Machine Learning (DML) to estimate the causal effect of gender cues on stereotype-consistent translation output. Our results reveal a striking asymmetry: cue direction has a large and statistically significant effect on translation outcomes, while cue presence alone produces virtually no effect. Even without any gender signal, MT systems default to stereotype-consistent translations in 62.9% of cases across three systems (DeepL, Google Translate, OpenAI). By leveraging this typological property, our causal analysis reveals that gender bias in contemporary MT and LLM-based translation systems runs deeper than surface-level cue processing, persisting as an embedded prior independent of any explicit gender signal in the input.</abstract>
<identifier type="citekey">albayrak-2026-winotr</identifier>
<location>
<url>https://aclanthology.org/2026.gitt-1.9/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>100</start>
<end>107</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T WinoTR: Evaluating Gender Bias in Machine Translation from a Gender-Neutral Language Using Causal Inference
%A Albayrak, Deniz
%Y Lardelli, Manuel
%Y Savoldi, Beatrice
%Y Hackenbuchner, Janiça
%Y Bentivogli, Luisa
%Y Gkovedarou, Eleni
%Y Daems, Joke
%S Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, the Netherlands
%F albayrak-2026-winotr
%X We present WinoTR, a Turkish adaptation of the WinoMT challenge dataset (Stanovsky et al., 2019). While WinoMT has been widely studied across multiple languages, its adaptation to Turkish — a morphologically rich language with no grammatical gender — and its analysis through a causal lens remain unexplored. Using 4,752 sentences adapted from the original dataset across pro-stereotypical, anti-stereotypical, and neutral conditions, we apply Double Machine Learning (DML) to estimate the causal effect of gender cues on stereotype-consistent translation output. Our results reveal a striking asymmetry: cue direction has a large and statistically significant effect on translation outcomes, while cue presence alone produces virtually no effect. Even without any gender signal, MT systems default to stereotype-consistent translations in 62.9% of cases across three systems (DeepL, Google Translate, OpenAI). By leveraging this typological property, our causal analysis reveals that gender bias in contemporary MT and LLM-based translation systems runs deeper than surface-level cue processing, persisting as an embedded prior independent of any explicit gender signal in the input.
%U https://aclanthology.org/2026.gitt-1.9/
%P 100-107
Markdown (Informal)
[WinoTR: Evaluating Gender Bias in Machine Translation from a Gender-Neutral Language Using Causal Inference](https://aclanthology.org/2026.gitt-1.9/) (Albayrak, GITT 2026)
ACL