@inproceedings{mamo-nigatu-2026-yeswa,
title = "Yeswa-Stories: A Three-Way Parallel Dataset of Female {A}frican Figures in Low-Web Data Languages",
author = "Mamo, Bethelhem and
Nigatu, Hellina",
editor = "Lardelli, Manuel and
Savoldi, Beatrice and
Hackenbuchner, Jani{\c{c}}a and
Bentivogli, Luisa and
Gkovedarou, Eleni and
Daems, Joke",
booktitle = "Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies ({GITT} 2026)",
month = jun,
year = "2026",
address = "Tilburg, the Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.gitt-1.7/",
pages = "81--86",
abstract = "Language technologies used in everyday settings such as machine translation systems risk perpetuating societal bias. As previous work shows, biases in these systems not only underscore representational harm but also materialize into economic disparities in resources required to correct errors for the disadvantaged social group. Prior work in creating benchmarks for gender bias in machine translation systems 1) focus primarily on high-resourced language pairs or a low-resourced language paired with a high resource language, 2) use template based benchmarks that usually focus on occupational biases and stereotypes, and 3) translate high-resource benchmarks which may lack cultural significance to low-resourced languages. In this paper, we introduce Yeswa-Stories, a three-way parallel dataset comprising 1,300 aligned sentences in Amharic, Afaan Oromo, and Tigrinya. The dataset focuses on narratives about women and is designed to support research on gender representation in translation. We constructed the dataset in two ways: first, we collected English sentences from Wikipedia articles about notable African women and translated them into the three target languages using human translators. To improve cultural representativeness, we further augment the dataset with locally sourced content reflecting the cultural context where the languages are spoken. Our dataset contributes a new resource for studying gender-inclusive translation in low-resourced settings."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mamo-nigatu-2026-yeswa">
<titleInfo>
<title>Yeswa-Stories: A Three-Way Parallel Dataset of Female African Figures in Low-Web Data Languages</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bethelhem</namePart>
<namePart type="family">Mamo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hellina</namePart>
<namePart type="family">Nigatu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Beatrice</namePart>
<namePart type="family">Savoldi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luisa</namePart>
<namePart type="family">Bentivogli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eleni</namePart>
<namePart type="family">Gkovedarou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joke</namePart>
<namePart type="family">Daems</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, the Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Language technologies used in everyday settings such as machine translation systems risk perpetuating societal bias. As previous work shows, biases in these systems not only underscore representational harm but also materialize into economic disparities in resources required to correct errors for the disadvantaged social group. Prior work in creating benchmarks for gender bias in machine translation systems 1) focus primarily on high-resourced language pairs or a low-resourced language paired with a high resource language, 2) use template based benchmarks that usually focus on occupational biases and stereotypes, and 3) translate high-resource benchmarks which may lack cultural significance to low-resourced languages. In this paper, we introduce Yeswa-Stories, a three-way parallel dataset comprising 1,300 aligned sentences in Amharic, Afaan Oromo, and Tigrinya. The dataset focuses on narratives about women and is designed to support research on gender representation in translation. We constructed the dataset in two ways: first, we collected English sentences from Wikipedia articles about notable African women and translated them into the three target languages using human translators. To improve cultural representativeness, we further augment the dataset with locally sourced content reflecting the cultural context where the languages are spoken. Our dataset contributes a new resource for studying gender-inclusive translation in low-resourced settings.</abstract>
<identifier type="citekey">mamo-nigatu-2026-yeswa</identifier>
<location>
<url>https://aclanthology.org/2026.gitt-1.7/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>81</start>
<end>86</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Yeswa-Stories: A Three-Way Parallel Dataset of Female African Figures in Low-Web Data Languages
%A Mamo, Bethelhem
%A Nigatu, Hellina
%Y Lardelli, Manuel
%Y Savoldi, Beatrice
%Y Hackenbuchner, Janiça
%Y Bentivogli, Luisa
%Y Gkovedarou, Eleni
%Y Daems, Joke
%S Proceedings of the 4th Workshop on Gender-Inclusive Translation Technologies (GITT 2026)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, the Netherlands
%F mamo-nigatu-2026-yeswa
%X Language technologies used in everyday settings such as machine translation systems risk perpetuating societal bias. As previous work shows, biases in these systems not only underscore representational harm but also materialize into economic disparities in resources required to correct errors for the disadvantaged social group. Prior work in creating benchmarks for gender bias in machine translation systems 1) focus primarily on high-resourced language pairs or a low-resourced language paired with a high resource language, 2) use template based benchmarks that usually focus on occupational biases and stereotypes, and 3) translate high-resource benchmarks which may lack cultural significance to low-resourced languages. In this paper, we introduce Yeswa-Stories, a three-way parallel dataset comprising 1,300 aligned sentences in Amharic, Afaan Oromo, and Tigrinya. The dataset focuses on narratives about women and is designed to support research on gender representation in translation. We constructed the dataset in two ways: first, we collected English sentences from Wikipedia articles about notable African women and translated them into the three target languages using human translators. To improve cultural representativeness, we further augment the dataset with locally sourced content reflecting the cultural context where the languages are spoken. Our dataset contributes a new resource for studying gender-inclusive translation in low-resourced settings.
%U https://aclanthology.org/2026.gitt-1.7/
%P 81-86
Markdown (Informal)
[Yeswa-Stories: A Three-Way Parallel Dataset of Female African Figures in Low-Web Data Languages](https://aclanthology.org/2026.gitt-1.7/) (Mamo & Nigatu, GITT 2026)
ACL