@inproceedings{pandeiro-etal-2026-multilingual,
title = "A Multilingual Red Teaming{--}Driven Safety Analysis of {LLM}s",
author = "Pandeiro, Patr{\'i}cia and
Cabarr{\~a}o, Vera and
Moniz, Helena",
editor = "Shterionov, Dimitar and
Vanmassenhove, Eva and
De Sisto, Mirella and
Blain, Fred and
Pourmostafa Roshan Sharami, Javad and
Lepp, Lisa and
Manna, Chiara and
Rescigno, Argentina Anna and
Karakanta, Alina and
Rigouts Terryn, Ayla and
Lardelli, Manuel and
Resende, Natalia and
Murgolo, Elena and
Hackenbuchner, Jani{\c{c}}a and
Zaretskaya, Anna and
Espl{\`a}-Gomis, Miquel and
Etchegoyhen, Thierry and
Gromann, Dagmar and
Bawden, Rachel and
Haddow, Barry and
Szoc, Sara and
Forcada, Mikel and
Moniz, Helena",
booktitle = "Proceedings of the 26th Annual Conference of the {E}uropean Association for Machine Translation (Volume 1)",
month = jun,
year = "2026",
address = "Tilburg, The Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.eamt-1.46/",
pages = "733--743",
ISBN = "9789403901411",
abstract = "This work benchmarks safety across several large language models (LLMs) and compares their performances through red teaming, which simulates adversarial attacks and identifies vulnerabilities in the systems. Using two public datasets and a proprietary dataset, the models were tested with three purposes. First, a red teaming test was conducted to establish a safety comparison between five models in English and Portuguese. The results revealed that, in general, Sugarloaf 3.1 is the safest model, but that Vesuvius 4.0 slightly outperforms it in Portuguese, also revealing that both outperform GPT-4o. Afterwards, three models were tested with one guardrailing prompt, that encourages safe interactions, and two content moderation prompts, in both languages, to understand the strengths of the current guardrails, as well as the effectiveness of the content moderation task. The results show that current guardrails are sufficient, notwithstanding room for improvement (particularly for Portuguese), but that the performance of the content moderation task was substandard, even for the best performing model {--} GPT-4o. Finally, the 3.0 TowerLLM models were tested in English to evaluate the effect that tokens and temperature have on the output, revealing that an intermediate token limit leads to safer responses while a higher temperature causes performance degradation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="pandeiro-etal-2026-multilingual">
<titleInfo>
<title>A Multilingual Red Teaming–Driven Safety Analysis of LLMs</title>
</titleInfo>
<name type="personal">
<namePart type="given">Patrícia</namePart>
<namePart type="family">Pandeiro</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vera</namePart>
<namePart type="family">Cabarrão</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitar</namePart>
<namePart type="family">Shterionov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mirella</namePart>
<namePart type="family">De Sisto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fred</namePart>
<namePart type="family">Blain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javad</namePart>
<namePart type="family">Pourmostafa Roshan Sharami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Lepp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Karakanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="family">Rigouts Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Resende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Murgolo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Zaretskaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Miquel</namePart>
<namePart type="family">Esplà-Gomis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thierry</namePart>
<namePart type="family">Etchegoyhen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dagmar</namePart>
<namePart type="family">Gromann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barry</namePart>
<namePart type="family">Haddow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Szoc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikel</namePart>
<namePart type="family">Forcada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, The Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">9789403901411</identifier>
</relatedItem>
<abstract>This work benchmarks safety across several large language models (LLMs) and compares their performances through red teaming, which simulates adversarial attacks and identifies vulnerabilities in the systems. Using two public datasets and a proprietary dataset, the models were tested with three purposes. First, a red teaming test was conducted to establish a safety comparison between five models in English and Portuguese. The results revealed that, in general, Sugarloaf 3.1 is the safest model, but that Vesuvius 4.0 slightly outperforms it in Portuguese, also revealing that both outperform GPT-4o. Afterwards, three models were tested with one guardrailing prompt, that encourages safe interactions, and two content moderation prompts, in both languages, to understand the strengths of the current guardrails, as well as the effectiveness of the content moderation task. The results show that current guardrails are sufficient, notwithstanding room for improvement (particularly for Portuguese), but that the performance of the content moderation task was substandard, even for the best performing model – GPT-4o. Finally, the 3.0 TowerLLM models were tested in English to evaluate the effect that tokens and temperature have on the output, revealing that an intermediate token limit leads to safer responses while a higher temperature causes performance degradation.</abstract>
<identifier type="citekey">pandeiro-etal-2026-multilingual</identifier>
<location>
<url>https://aclanthology.org/2026.eamt-1.46/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>733</start>
<end>743</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Multilingual Red Teaming–Driven Safety Analysis of LLMs
%A Pandeiro, Patrícia
%A Cabarrão, Vera
%A Moniz, Helena
%Y Shterionov, Dimitar
%Y Vanmassenhove, Eva
%Y De Sisto, Mirella
%Y Blain, Fred
%Y Pourmostafa Roshan Sharami, Javad
%Y Lepp, Lisa
%Y Manna, Chiara
%Y Rescigno, Argentina Anna
%Y Karakanta, Alina
%Y Rigouts Terryn, Ayla
%Y Lardelli, Manuel
%Y Resende, Natalia
%Y Murgolo, Elena
%Y Hackenbuchner, Janiça
%Y Zaretskaya, Anna
%Y Esplà-Gomis, Miquel
%Y Etchegoyhen, Thierry
%Y Gromann, Dagmar
%Y Bawden, Rachel
%Y Haddow, Barry
%Y Szoc, Sara
%Y Forcada, Mikel
%Y Moniz, Helena
%S Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, The Netherlands
%@ 9789403901411
%F pandeiro-etal-2026-multilingual
%X This work benchmarks safety across several large language models (LLMs) and compares their performances through red teaming, which simulates adversarial attacks and identifies vulnerabilities in the systems. Using two public datasets and a proprietary dataset, the models were tested with three purposes. First, a red teaming test was conducted to establish a safety comparison between five models in English and Portuguese. The results revealed that, in general, Sugarloaf 3.1 is the safest model, but that Vesuvius 4.0 slightly outperforms it in Portuguese, also revealing that both outperform GPT-4o. Afterwards, three models were tested with one guardrailing prompt, that encourages safe interactions, and two content moderation prompts, in both languages, to understand the strengths of the current guardrails, as well as the effectiveness of the content moderation task. The results show that current guardrails are sufficient, notwithstanding room for improvement (particularly for Portuguese), but that the performance of the content moderation task was substandard, even for the best performing model – GPT-4o. Finally, the 3.0 TowerLLM models were tested in English to evaluate the effect that tokens and temperature have on the output, revealing that an intermediate token limit leads to safer responses while a higher temperature causes performance degradation.
%U https://aclanthology.org/2026.eamt-1.46/
%P 733-743
Markdown (Informal)
[A Multilingual Red Teaming–Driven Safety Analysis of LLMs](https://aclanthology.org/2026.eamt-1.46/) (Pandeiro et al., EAMT 2026)
ACL
- Patrícia Pandeiro, Vera Cabarrão, and Helena Moniz. 2026. A Multilingual Red Teaming–Driven Safety Analysis of LLMs. In Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1), pages 733–743, Tilburg, The Netherlands. European Association for Machine Translation.