@inproceedings{manna-etal-2026-rethinking,
title = "Rethinking Gender Annotation for Bias Evaluation in Machine Translation: Can {LLM}s Improve Reliability?",
author = "Manna, Chiara and
Rescigno, Argentina Anna and
Vanmassenhove, Eva",
editor = "Lardelli, Manuel and
Savoldi, Beatrice and
Hackenbuchner, Jani{\c{c}}a and
Bentivogli, Luisa and
Gkovedarou, Eleni and
Daems, Joke",
booktitle = "Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies ({GITT} 2026)",
month = jun,
year = "2026",
address = "Tilburg, the Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.gitt-1.3/",
pages = "16--30",
abstract = "The assessment of gender bias in Machine Translation critically depends on reliable methods for identifying grammatical gender in system outputs. In this paper, we compare the automated WinoMT annotation pipeline, based on word alignment and morphological tagging, with an instruction-tuned LLM (Qwen3-8B) used to annotate grammatical gender in English{--}Italian translations. Both approaches achieve similar levels of agreement with a human-annotated gold standard, but exhibit distinct systematic weaknesses. The WinoMT pipeline is sensitive to alignment shifts and morphological tagging limitations, often resulting in indeterminate gender labels. The LLM, in contrast, tends to favour binary gender labels even when noun phrases are morphologically gender-invariant. A qualitative analysis of model outputs and reasoning traces further suggests a lack of stable generalization of grammatical gender rules, even when illustrative exemplars are provided for in-context learning. This highlights important methodological limitations in using LLMs as objective annotators for gender evaluation tasks."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="manna-etal-2026-rethinking">
<titleInfo>
<title>Rethinking Gender Annotation for Bias Evaluation in Machine Translation: Can LLMs Improve Reliability?</title>
</titleInfo>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Beatrice</namePart>
<namePart type="family">Savoldi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luisa</namePart>
<namePart type="family">Bentivogli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eleni</namePart>
<namePart type="family">Gkovedarou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joke</namePart>
<namePart type="family">Daems</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, the Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The assessment of gender bias in Machine Translation critically depends on reliable methods for identifying grammatical gender in system outputs. In this paper, we compare the automated WinoMT annotation pipeline, based on word alignment and morphological tagging, with an instruction-tuned LLM (Qwen3-8B) used to annotate grammatical gender in English–Italian translations. Both approaches achieve similar levels of agreement with a human-annotated gold standard, but exhibit distinct systematic weaknesses. The WinoMT pipeline is sensitive to alignment shifts and morphological tagging limitations, often resulting in indeterminate gender labels. The LLM, in contrast, tends to favour binary gender labels even when noun phrases are morphologically gender-invariant. A qualitative analysis of model outputs and reasoning traces further suggests a lack of stable generalization of grammatical gender rules, even when illustrative exemplars are provided for in-context learning. This highlights important methodological limitations in using LLMs as objective annotators for gender evaluation tasks.</abstract>
<identifier type="citekey">manna-etal-2026-rethinking</identifier>
<location>
<url>https://aclanthology.org/2026.gitt-1.3/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>16</start>
<end>30</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Rethinking Gender Annotation for Bias Evaluation in Machine Translation: Can LLMs Improve Reliability?
%A Manna, Chiara
%A Rescigno, Argentina Anna
%A Vanmassenhove, Eva
%Y Lardelli, Manuel
%Y Savoldi, Beatrice
%Y Hackenbuchner, Janiça
%Y Bentivogli, Luisa
%Y Gkovedarou, Eleni
%Y Daems, Joke
%S Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, the Netherlands
%F manna-etal-2026-rethinking
%X The assessment of gender bias in Machine Translation critically depends on reliable methods for identifying grammatical gender in system outputs. In this paper, we compare the automated WinoMT annotation pipeline, based on word alignment and morphological tagging, with an instruction-tuned LLM (Qwen3-8B) used to annotate grammatical gender in English–Italian translations. Both approaches achieve similar levels of agreement with a human-annotated gold standard, but exhibit distinct systematic weaknesses. The WinoMT pipeline is sensitive to alignment shifts and morphological tagging limitations, often resulting in indeterminate gender labels. The LLM, in contrast, tends to favour binary gender labels even when noun phrases are morphologically gender-invariant. A qualitative analysis of model outputs and reasoning traces further suggests a lack of stable generalization of grammatical gender rules, even when illustrative exemplars are provided for in-context learning. This highlights important methodological limitations in using LLMs as objective annotators for gender evaluation tasks.
%U https://aclanthology.org/2026.gitt-1.3/
%P 16-30
Markdown (Informal)
[Rethinking Gender Annotation for Bias Evaluation in Machine Translation: Can LLMs Improve Reliability?](https://aclanthology.org/2026.gitt-1.3/) (Manna et al., GITT 2026)
ACL