@inproceedings{mahadeshwar-etal-2026-evaluating,
title = "Evaluating the Impact of Source Diversity for {RAG} in Historical Research",
author = "Mahadeshwar, Ruhi and
van Cranenburgh, Andreas and
Caselli, Tommaso and
Nissim, Malvina",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.53/",
doi = "10.63317/4yz9z3uvzasd",
pages = "716--734",
abstract = "Historical research increasingly benefits from large language models (LLMs). However, LLMs are prone to factual inaccuracy, unreliability, and biased interpretations of data. Retrieval-augmented generation (RAG) approaches have emerged as solutions, but may inadvertently perpetuate biased perspectives embedded in historical archives. This paper investigates how source diversity in RAG impacts perspective variation in historical question answering. We compile a multilingual corpus (English, French, Dutch) of historical documents spanning multiple countries and focus on Napoleon Bonaparte. We evaluate three Qwen3 models across ten questions using a multi-layered framework combining traditional metrics (BERTScore, ROUGE-L), frame semantics analysis, and syntactic profiling. Our results highlight that, while traditional similarity metrics suggest high semantic consistency, frame-semantic analysis exposes substantial perspective shifts. Baseline answers present ``flattened'' cross-lingual perspectives, whereas RAG introduces diversity. Critically, this diversity manifests differently across languages, demonstrating language-specific patterns. Our findings highlight limitations of traditional evaluation metrics for perspective-sensitive tasks and demonstrate that RAG constitutes active perspective transformation rather than neutral augmentation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mahadeshwar-etal-2026-evaluating">
<titleInfo>
<title>Evaluating the Impact of Source Diversity for RAG in Historical Research</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruhi</namePart>
<namePart type="family">Mahadeshwar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">van Cranenburgh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tommaso</namePart>
<namePart type="family">Caselli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Malvina</namePart>
<namePart type="family">Nissim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Historical research increasingly benefits from large language models (LLMs). However, LLMs are prone to factual inaccuracy, unreliability, and biased interpretations of data. Retrieval-augmented generation (RAG) approaches have emerged as solutions, but may inadvertently perpetuate biased perspectives embedded in historical archives. This paper investigates how source diversity in RAG impacts perspective variation in historical question answering. We compile a multilingual corpus (English, French, Dutch) of historical documents spanning multiple countries and focus on Napoleon Bonaparte. We evaluate three Qwen3 models across ten questions using a multi-layered framework combining traditional metrics (BERTScore, ROUGE-L), frame semantics analysis, and syntactic profiling. Our results highlight that, while traditional similarity metrics suggest high semantic consistency, frame-semantic analysis exposes substantial perspective shifts. Baseline answers present “flattened” cross-lingual perspectives, whereas RAG introduces diversity. Critically, this diversity manifests differently across languages, demonstrating language-specific patterns. Our findings highlight limitations of traditional evaluation metrics for perspective-sensitive tasks and demonstrate that RAG constitutes active perspective transformation rather than neutral augmentation.</abstract>
<identifier type="citekey">mahadeshwar-etal-2026-evaluating</identifier>
<identifier type="doi">10.63317/4yz9z3uvzasd</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.53/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>716</start>
<end>734</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating the Impact of Source Diversity for RAG in Historical Research
%A Mahadeshwar, Ruhi
%A van Cranenburgh, Andreas
%A Caselli, Tommaso
%A Nissim, Malvina
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F mahadeshwar-etal-2026-evaluating
%X Historical research increasingly benefits from large language models (LLMs). However, LLMs are prone to factual inaccuracy, unreliability, and biased interpretations of data. Retrieval-augmented generation (RAG) approaches have emerged as solutions, but may inadvertently perpetuate biased perspectives embedded in historical archives. This paper investigates how source diversity in RAG impacts perspective variation in historical question answering. We compile a multilingual corpus (English, French, Dutch) of historical documents spanning multiple countries and focus on Napoleon Bonaparte. We evaluate three Qwen3 models across ten questions using a multi-layered framework combining traditional metrics (BERTScore, ROUGE-L), frame semantics analysis, and syntactic profiling. Our results highlight that, while traditional similarity metrics suggest high semantic consistency, frame-semantic analysis exposes substantial perspective shifts. Baseline answers present “flattened” cross-lingual perspectives, whereas RAG introduces diversity. Critically, this diversity manifests differently across languages, demonstrating language-specific patterns. Our findings highlight limitations of traditional evaluation metrics for perspective-sensitive tasks and demonstrate that RAG constitutes active perspective transformation rather than neutral augmentation.
%R 10.63317/4yz9z3uvzasd
%U https://aclanthology.org/2026.lrec-1.53/
%U https://doi.org/10.63317/4yz9z3uvzasd
%P 716-734
Markdown (Informal)
[Evaluating the Impact of Source Diversity for RAG in Historical Research](https://aclanthology.org/2026.lrec-1.53/) (Mahadeshwar et al., LREC 2026)
ACL