@inproceedings{faheem-etal-2026-beyond,
title = "Beyond Fine-Tuning: {P}rocrustes Alignment of Multilingual Embeddings for Low-Resource Cross-Lingual Retrieval",
author = "Faheem, Ali and
Hammad, Muhammad and
Ullah, Faizad and
Hassan, Ahmed and
Rasool, Fezan and
Karim, Asim",
editor = "Ojha, Atul Kr. and
Sakti, Sakriani and
Soria, Claudia and
Melero, Maite and
McCrae, John P. and
Lignos, Constantine and
Liu, Chao-Hong and
Claramunt, German Rigau and
Rehm, Georg",
booktitle = "Proceedings of the {SIGUL} 2026 Joint Workshop with {ELE}, {EURALI}, and {DCLRL}: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.sigul-1.23/",
doi = "10.63317/3bfgv7a4e3xh",
pages = "233--241",
abstract = "Multilingual sentence-embedding models are widely used for cross-lingual retrieval; however, their performance drops significantly in low-resource languages. The Urdu language, which is considered a low-resource language by the NL community, poses this challenge, despite being spoken by over 246 million people worldwide. Its distribution in training corpora results in poor alignment with English within shared embedding spaces. To resolve this misalignment without model fine-tuning, we apply Procrustes transformation, which is an orthogonal post-hoc alignment method with a closed-form solution. We utilize SQuAD and UQA datasets to learn a rotation matrix from a small set of sentence pairs and evaluate its effect across five multilingual embedding models (MiniLM, DistilUSE, E5-Base, LaBSE, and E5-Large) and perform geometric alignment, cross-lingual retrieval, and question-answering tasks on these models. We find that cosine distances between parallel pairs decrease by up to 38.67{\%}, and retrieval accuracy improves by 12.49{\%} points in Recall@1. We also analyze that models with better pre-trained cross-lingual representations exhibit a saturation effect, showing minimal retrieval change even as geometric tightening increases. Our error analysis reveals that morphologically complex queries and colloquial expressions remain challenging, indicating representational limitations beyond the scope of a linear transformation. These findings demonstrate that a computationally inexpensive alignment step can meaningfully improve cross-lingual retrieval for low-resource languages, with implications for retrieval-augmented generation (RAG) in resource-constrained settings."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="faheem-etal-2026-beyond">
<titleInfo>
<title>Beyond Fine-Tuning: Procrustes Alignment of Multilingual Embeddings for Low-Resource Cross-Lingual Retrieval</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ali</namePart>
<namePart type="family">Faheem</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Muhammad</namePart>
<namePart type="family">Hammad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Faizad</namePart>
<namePart type="family">Ullah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahmed</namePart>
<namePart type="family">Hassan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fezan</namePart>
<namePart type="family">Rasool</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Asim</namePart>
<namePart type="family">Karim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages</title>
</titleInfo>
<name type="personal">
<namePart type="given">Atul</namePart>
<namePart type="given">Kr.</namePart>
<namePart type="family">Ojha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sakriani</namePart>
<namePart type="family">Sakti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Claudia</namePart>
<namePart type="family">Soria</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maite</namePart>
<namePart type="family">Melero</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">John</namePart>
<namePart type="given">P</namePart>
<namePart type="family">McCrae</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Constantine</namePart>
<namePart type="family">Lignos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chao-Hong</namePart>
<namePart type="family">Liu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="given">Rigau</namePart>
<namePart type="family">Claramunt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Multilingual sentence-embedding models are widely used for cross-lingual retrieval; however, their performance drops significantly in low-resource languages. The Urdu language, which is considered a low-resource language by the NL community, poses this challenge, despite being spoken by over 246 million people worldwide. Its distribution in training corpora results in poor alignment with English within shared embedding spaces. To resolve this misalignment without model fine-tuning, we apply Procrustes transformation, which is an orthogonal post-hoc alignment method with a closed-form solution. We utilize SQuAD and UQA datasets to learn a rotation matrix from a small set of sentence pairs and evaluate its effect across five multilingual embedding models (MiniLM, DistilUSE, E5-Base, LaBSE, and E5-Large) and perform geometric alignment, cross-lingual retrieval, and question-answering tasks on these models. We find that cosine distances between parallel pairs decrease by up to 38.67%, and retrieval accuracy improves by 12.49% points in Recall@1. We also analyze that models with better pre-trained cross-lingual representations exhibit a saturation effect, showing minimal retrieval change even as geometric tightening increases. Our error analysis reveals that morphologically complex queries and colloquial expressions remain challenging, indicating representational limitations beyond the scope of a linear transformation. These findings demonstrate that a computationally inexpensive alignment step can meaningfully improve cross-lingual retrieval for low-resource languages, with implications for retrieval-augmented generation (RAG) in resource-constrained settings.</abstract>
<identifier type="citekey">faheem-etal-2026-beyond</identifier>
<identifier type="doi">10.63317/3bfgv7a4e3xh</identifier>
<location>
<url>https://aclanthology.org/2026.sigul-1.23/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>233</start>
<end>241</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Beyond Fine-Tuning: Procrustes Alignment of Multilingual Embeddings for Low-Resource Cross-Lingual Retrieval
%A Faheem, Ali
%A Hammad, Muhammad
%A Ullah, Faizad
%A Hassan, Ahmed
%A Rasool, Fezan
%A Karim, Asim
%Y Ojha, Atul Kr.
%Y Sakti, Sakriani
%Y Soria, Claudia
%Y Melero, Maite
%Y McCrae, John P.
%Y Lignos, Constantine
%Y Liu, Chao-Hong
%Y Claramunt, German Rigau
%Y Rehm, Georg
%S Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca, Spain
%F faheem-etal-2026-beyond
%X Multilingual sentence-embedding models are widely used for cross-lingual retrieval; however, their performance drops significantly in low-resource languages. The Urdu language, which is considered a low-resource language by the NL community, poses this challenge, despite being spoken by over 246 million people worldwide. Its distribution in training corpora results in poor alignment with English within shared embedding spaces. To resolve this misalignment without model fine-tuning, we apply Procrustes transformation, which is an orthogonal post-hoc alignment method with a closed-form solution. We utilize SQuAD and UQA datasets to learn a rotation matrix from a small set of sentence pairs and evaluate its effect across five multilingual embedding models (MiniLM, DistilUSE, E5-Base, LaBSE, and E5-Large) and perform geometric alignment, cross-lingual retrieval, and question-answering tasks on these models. We find that cosine distances between parallel pairs decrease by up to 38.67%, and retrieval accuracy improves by 12.49% points in Recall@1. We also analyze that models with better pre-trained cross-lingual representations exhibit a saturation effect, showing minimal retrieval change even as geometric tightening increases. Our error analysis reveals that morphologically complex queries and colloquial expressions remain challenging, indicating representational limitations beyond the scope of a linear transformation. These findings demonstrate that a computationally inexpensive alignment step can meaningfully improve cross-lingual retrieval for low-resource languages, with implications for retrieval-augmented generation (RAG) in resource-constrained settings.
%R 10.63317/3bfgv7a4e3xh
%U https://aclanthology.org/2026.sigul-1.23/
%U https://doi.org/10.63317/3bfgv7a4e3xh
%P 233-241
Markdown (Informal)
[Beyond Fine-Tuning: Procrustes Alignment of Multilingual Embeddings for Low-Resource Cross-Lingual Retrieval](https://aclanthology.org/2026.sigul-1.23/) (Faheem et al., SIGUL-EURALI-DCLRL 2026)
ACL
- Ali Faheem, Muhammad Hammad, Faizad Ullah, Ahmed Hassan, Fezan Rasool, and Asim Karim. 2026. Beyond Fine-Tuning: Procrustes Alignment of Multilingual Embeddings for Low-Resource Cross-Lingual Retrieval. In Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages, pages 233–241, Palma, Mallorca, Spain. ELRA Language Resources Association (ELRA).