@inproceedings{tsukagoshi-ohmukai-2026-aligned,
title = "Aligned Parallel Corpus of the {V}edic Saṁhit{\={a}}s for Machine Translation",
author = "Tsukagoshi, Yuzuki and
Ohmukai, Ikki",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.272/",
doi = "10.63317/3faoikwftvwt",
pages = "3434--3444",
abstract = "We introduce a verse-/paragraph-aligned parallel corpus for three Vedic Saṁhit{\={a}}s {--}the R̥gveda (R̥V), the Atharvaveda {\'S}aunaka (AV{\'S}), and the Taittir{\={i}}ya Saṁhit{\={a}} (TS){--} paired with authoritative public-domain translations (Geldner for R̥V, Whitney for AV{\'S}, and Keith for TS). The source texts are drawn from established digital editions (e.g., TITUS and VedaWeb) and normalized under ISO 15919. Each Sanskrit segment is aligned to exactly one translated unit (verse or paragraph for TS prose), yielding a unified, model-ready format. Using this resource, we fine-tune and evaluate three large language models {--}GPT-4.1 nano, Gemini 2.5 Flash, and Mitra{--} on Vedic$\to$German/English translation. Evaluation combines surface and semantic metrics (case-insensitive sacreBLEU and COMET), enabling a balanced assessment of form and meaning. Results show consistent in-domain gains after supervised fine-tuning, but substantial cross-domain degradation when models are tested on unseen Saṁhit{\={a}}s, indicating pronounced stylistic and lexical divergence among R̥V, AV{\'S}, and TS. These findings motivate domain-aware training and reporting practices for Vedic machine translation. We release the corpus with standardized splits and preprocessing to support reproducibility and future d research on historical language modeling, alignment, and translation for low-resource ancient languages."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tsukagoshi-ohmukai-2026-aligned">
<titleInfo>
<title>Aligned Parallel Corpus of the Vedic Saṁhitās for Machine Translation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Yuzuki</namePart>
<namePart type="family">Tsukagoshi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ikki</namePart>
<namePart type="family">Ohmukai</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We introduce a verse-/paragraph-aligned parallel corpus for three Vedic Saṁhitās –the R̥gveda (R̥V), the Atharvaveda Śaunaka (AVŚ), and the Taittirīya Saṁhitā (TS)– paired with authoritative public-domain translations (Geldner for R̥V, Whitney for AVŚ, and Keith for TS). The source texts are drawn from established digital editions (e.g., TITUS and VedaWeb) and normalized under ISO 15919. Each Sanskrit segment is aligned to exactly one translated unit (verse or paragraph for TS prose), yielding a unified, model-ready format. Using this resource, we fine-tune and evaluate three large language models –GPT-4.1 nano, Gemini 2.5 Flash, and Mitra– on VedicGerman/English translation. Evaluation combines surface and semantic metrics (case-insensitive sacreBLEU and COMET), enabling a balanced assessment of form and meaning. Results show consistent in-domain gains after supervised fine-tuning, but substantial cross-domain degradation when models are tested on unseen Saṁhitās, indicating pronounced stylistic and lexical divergence among R̥V, AVŚ, and TS. These findings motivate domain-aware training and reporting practices for Vedic machine translation. We release the corpus with standardized splits and preprocessing to support reproducibility and future d research on historical language modeling, alignment, and translation for low-resource ancient languages.</abstract>
<identifier type="citekey">tsukagoshi-ohmukai-2026-aligned</identifier>
<identifier type="doi">10.63317/3faoikwftvwt</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.272/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>3434</start>
<end>3444</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Aligned Parallel Corpus of the Vedic Saṁhitās for Machine Translation
%A Tsukagoshi, Yuzuki
%A Ohmukai, Ikki
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F tsukagoshi-ohmukai-2026-aligned
%X We introduce a verse-/paragraph-aligned parallel corpus for three Vedic Saṁhitās –the R̥gveda (R̥V), the Atharvaveda Śaunaka (AVŚ), and the Taittirīya Saṁhitā (TS)– paired with authoritative public-domain translations (Geldner for R̥V, Whitney for AVŚ, and Keith for TS). The source texts are drawn from established digital editions (e.g., TITUS and VedaWeb) and normalized under ISO 15919. Each Sanskrit segment is aligned to exactly one translated unit (verse or paragraph for TS prose), yielding a unified, model-ready format. Using this resource, we fine-tune and evaluate three large language models –GPT-4.1 nano, Gemini 2.5 Flash, and Mitra– on VedicGerman/English translation. Evaluation combines surface and semantic metrics (case-insensitive sacreBLEU and COMET), enabling a balanced assessment of form and meaning. Results show consistent in-domain gains after supervised fine-tuning, but substantial cross-domain degradation when models are tested on unseen Saṁhitās, indicating pronounced stylistic and lexical divergence among R̥V, AVŚ, and TS. These findings motivate domain-aware training and reporting practices for Vedic machine translation. We release the corpus with standardized splits and preprocessing to support reproducibility and future d research on historical language modeling, alignment, and translation for low-resource ancient languages.
%R 10.63317/3faoikwftvwt
%U https://aclanthology.org/2026.lrec-1.272/
%U https://doi.org/10.63317/3faoikwftvwt
%P 3434-3444
Markdown (Informal)
[Aligned Parallel Corpus of the Vedic Saṁhitās for Machine Translation](https://aclanthology.org/2026.lrec-1.272/) (Tsukagoshi & Ohmukai, LREC 2026)
ACL