@inproceedings{jotie-mitra-2026-improving,
title = "Improving {A}mharic Information Retrieval with Translative and Multi-Agent Debate Retrieval Augmented Generation",
author = "Jotie, Abel Alemu and
Mitra, Prasenjit",
editor = "Matfunjwa, Muzi and
Setaka, Mmasibidi and
Mabuya, Rooweither and
van Zaanen, Menno",
booktitle = "Proceedings of Resources for {A}frican Indigenous Languages ({RAIL}) 2026 @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.rail-1.6/",
doi = "10.63317/3hbm9z4okcrx",
pages = "52--61",
abstract = "Retrieval-augmented generation (RAG) has been used to improve the accuracy and transparency of outputs produced by large language models (LLMs) by integrating external knowledge; however, applying RAG to low-resource languages presents unique challenges, including poor embedding representations, low retrieval quality, and semantic gaps caused by the scarcity of digital documents. In this research, we address these challenges for a selected low-resource language, Amharic, by using translative and debate-based RAG techniques to improve retrieval and reasoning. This paper outlines the key problems and research gaps in applying RAG to low-resource languages and introduces a method to enhance RAG performance for Amharic. Additionally, we introduce the first comprehensive \textbf{A}mharic \textbf{R}etrieval-Augmented \textbf{G}eneration \textbf{B}enchmark (ARGB), designed to capture grammatical, cultural, and writing-system-specific constraints of the Amharic language. ARGB evaluates not only retrieval and generation quality, but also noise robustness, counterfactual robustness, negative rejection, and multi-source information integration, providing a holistic assessment of RAG capabilities. The dataset, which spans a wide range of categories, is evaluated using multiple evaluation metrics. Furthermore, we demonstrate that, using our dataset, translation-based and debate-based methods substantially improve various aspects of RAG pipeline assessment in the Amharic language. This work aims to improve the reliability, accessibility, and inclusiveness of AI systems for Amharic speakers while providing a scalable framework for other low-resource languages. Current progress on the code and benchmark can be found on this GitHub link: \href{https://anonymous.4open.science/r/AmharicRAG-FC34}{link}."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="jotie-mitra-2026-improving">
<titleInfo>
<title>Improving Amharic Information Retrieval with Translative and Multi-Agent Debate Retrieval Augmented Generation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Abel</namePart>
<namePart type="given">Alemu</namePart>
<namePart type="family">Jotie</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Prasenjit</namePart>
<namePart type="family">Mitra</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Resources for African Indigenous Languages (RAIL) 2026 @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Muzi</namePart>
<namePart type="family">Matfunjwa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mmasibidi</namePart>
<namePart type="family">Setaka</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rooweither</namePart>
<namePart type="family">Mabuya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Menno</namePart>
<namePart type="family">van Zaanen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Retrieval-augmented generation (RAG) has been used to improve the accuracy and transparency of outputs produced by large language models (LLMs) by integrating external knowledge; however, applying RAG to low-resource languages presents unique challenges, including poor embedding representations, low retrieval quality, and semantic gaps caused by the scarcity of digital documents. In this research, we address these challenges for a selected low-resource language, Amharic, by using translative and debate-based RAG techniques to improve retrieval and reasoning. This paper outlines the key problems and research gaps in applying RAG to low-resource languages and introduces a method to enhance RAG performance for Amharic. Additionally, we introduce the first comprehensive Amharic Retrieval-Augmented Generation Benchmark (ARGB), designed to capture grammatical, cultural, and writing-system-specific constraints of the Amharic language. ARGB evaluates not only retrieval and generation quality, but also noise robustness, counterfactual robustness, negative rejection, and multi-source information integration, providing a holistic assessment of RAG capabilities. The dataset, which spans a wide range of categories, is evaluated using multiple evaluation metrics. Furthermore, we demonstrate that, using our dataset, translation-based and debate-based methods substantially improve various aspects of RAG pipeline assessment in the Amharic language. This work aims to improve the reliability, accessibility, and inclusiveness of AI systems for Amharic speakers while providing a scalable framework for other low-resource languages. Current progress on the code and benchmark can be found on this GitHub link: \hrefhttps://anonymous.4open.science/r/AmharicRAG-FC34link.</abstract>
<identifier type="citekey">jotie-mitra-2026-improving</identifier>
<identifier type="doi">10.63317/3hbm9z4okcrx</identifier>
<location>
<url>https://aclanthology.org/2026.rail-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>52</start>
<end>61</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Improving Amharic Information Retrieval with Translative and Multi-Agent Debate Retrieval Augmented Generation
%A Jotie, Abel Alemu
%A Mitra, Prasenjit
%Y Matfunjwa, Muzi
%Y Setaka, Mmasibidi
%Y Mabuya, Rooweither
%Y van Zaanen, Menno
%S Proceedings of Resources for African Indigenous Languages (RAIL) 2026 @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F jotie-mitra-2026-improving
%X Retrieval-augmented generation (RAG) has been used to improve the accuracy and transparency of outputs produced by large language models (LLMs) by integrating external knowledge; however, applying RAG to low-resource languages presents unique challenges, including poor embedding representations, low retrieval quality, and semantic gaps caused by the scarcity of digital documents. In this research, we address these challenges for a selected low-resource language, Amharic, by using translative and debate-based RAG techniques to improve retrieval and reasoning. This paper outlines the key problems and research gaps in applying RAG to low-resource languages and introduces a method to enhance RAG performance for Amharic. Additionally, we introduce the first comprehensive Amharic Retrieval-Augmented Generation Benchmark (ARGB), designed to capture grammatical, cultural, and writing-system-specific constraints of the Amharic language. ARGB evaluates not only retrieval and generation quality, but also noise robustness, counterfactual robustness, negative rejection, and multi-source information integration, providing a holistic assessment of RAG capabilities. The dataset, which spans a wide range of categories, is evaluated using multiple evaluation metrics. Furthermore, we demonstrate that, using our dataset, translation-based and debate-based methods substantially improve various aspects of RAG pipeline assessment in the Amharic language. This work aims to improve the reliability, accessibility, and inclusiveness of AI systems for Amharic speakers while providing a scalable framework for other low-resource languages. Current progress on the code and benchmark can be found on this GitHub link: \hrefhttps://anonymous.4open.science/r/AmharicRAG-FC34link.
%R 10.63317/3hbm9z4okcrx
%U https://aclanthology.org/2026.rail-1.6/
%U https://doi.org/10.63317/3hbm9z4okcrx
%P 52-61
Markdown (Informal)
[Improving Amharic Information Retrieval with Translative and Multi-Agent Debate Retrieval Augmented Generation](https://aclanthology.org/2026.rail-1.6/) (Jotie & Mitra, RAIL 2026)
ACL