@inproceedings{tian-wu-2026-benchmarking,
title = "Benchmarking Large Language Models for Game Localization Quality Assurance: A Cross-Model, Cross-Lingual Analysis",
author = "Tian, Mao and
Wu, Na",
editor = "Briakou, Eleftheria and
Gwinnup, Jeremy and
Goel, Shivali",
booktitle = "Proceedings of the 17th Conference of the Association for Machine Translation in the {A}mericas (Volume 1: Research Track)",
month = aug,
year = "2026",
address = "Qu{\'e}bec City, Canada",
publisher = "Association for Machine Translation in the Americas",
url = "https://aclanthology.org/2026.amta-research.12/",
pages = "186--201",
abstract = "Localization quality assurance (LQA) is a critical component of game development, where manual review of large volumes of translated text is time-consuming and costly. Recent advances in large language models (LLMs) suggest strong potential for automated LQA, yet their effectiveness across different models, target languages, and game domains remains insufficiently understood. We present a comprehensive benchmark evaluating eight LLMs, including both closed-source and open-weight models, on English-to-six-language gaming LQA tasks across two game genres. Our dataset comprises 96 evaluation settings with a total of 48,000 translation samples. The results show that Claude Sonnet 4 achieves the best overall performance (F1 = 0.766), followed by Qwen-2.5-72B (F1 = 0.711) and Gemini 2.0 Flash (F1 = 0.691). We observe that (1) the target language does not significantly affect model performance (p = 0.285), (2) models achieve their most consistent performance on French, while Japanese is the most challenging target language, and (3) game genre (RPG vs. strategy) has minimal impact on accuracy. While closed-source models achieve the highest overall performance, open-weight alternatives such as Qwen-2.5-72B provide competitive quality at substantially lower cost. These findings provide practical guidance for deploying LLM-based LQA systems in production game localization workflows."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tian-wu-2026-benchmarking">
<titleInfo>
<title>Benchmarking Large Language Models for Game Localization Quality Assurance: A Cross-Model, Cross-Lingual Analysis</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mao</namePart>
<namePart type="family">Tian</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Na</namePart>
<namePart type="family">Wu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Conference of the Association for Machine Translation in the Americas (Volume 1: Research Track)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Eleftheria</namePart>
<namePart type="family">Briakou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jeremy</namePart>
<namePart type="family">Gwinnup</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shivali</namePart>
<namePart type="family">Goel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Machine Translation in the Americas</publisher>
<place>
<placeTerm type="text">Québec City, Canada</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Localization quality assurance (LQA) is a critical component of game development, where manual review of large volumes of translated text is time-consuming and costly. Recent advances in large language models (LLMs) suggest strong potential for automated LQA, yet their effectiveness across different models, target languages, and game domains remains insufficiently understood. We present a comprehensive benchmark evaluating eight LLMs, including both closed-source and open-weight models, on English-to-six-language gaming LQA tasks across two game genres. Our dataset comprises 96 evaluation settings with a total of 48,000 translation samples. The results show that Claude Sonnet 4 achieves the best overall performance (F1 = 0.766), followed by Qwen-2.5-72B (F1 = 0.711) and Gemini 2.0 Flash (F1 = 0.691). We observe that (1) the target language does not significantly affect model performance (p = 0.285), (2) models achieve their most consistent performance on French, while Japanese is the most challenging target language, and (3) game genre (RPG vs. strategy) has minimal impact on accuracy. While closed-source models achieve the highest overall performance, open-weight alternatives such as Qwen-2.5-72B provide competitive quality at substantially lower cost. These findings provide practical guidance for deploying LLM-based LQA systems in production game localization workflows.</abstract>
<identifier type="citekey">tian-wu-2026-benchmarking</identifier>
<location>
<url>https://aclanthology.org/2026.amta-research.12/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>186</start>
<end>201</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Benchmarking Large Language Models for Game Localization Quality Assurance: A Cross-Model, Cross-Lingual Analysis
%A Tian, Mao
%A Wu, Na
%Y Briakou, Eleftheria
%Y Gwinnup, Jeremy
%Y Goel, Shivali
%S Proceedings of the 17th Conference of the Association for Machine Translation in the Americas (Volume 1: Research Track)
%D 2026
%8 August
%I Association for Machine Translation in the Americas
%C Québec City, Canada
%F tian-wu-2026-benchmarking
%X Localization quality assurance (LQA) is a critical component of game development, where manual review of large volumes of translated text is time-consuming and costly. Recent advances in large language models (LLMs) suggest strong potential for automated LQA, yet their effectiveness across different models, target languages, and game domains remains insufficiently understood. We present a comprehensive benchmark evaluating eight LLMs, including both closed-source and open-weight models, on English-to-six-language gaming LQA tasks across two game genres. Our dataset comprises 96 evaluation settings with a total of 48,000 translation samples. The results show that Claude Sonnet 4 achieves the best overall performance (F1 = 0.766), followed by Qwen-2.5-72B (F1 = 0.711) and Gemini 2.0 Flash (F1 = 0.691). We observe that (1) the target language does not significantly affect model performance (p = 0.285), (2) models achieve their most consistent performance on French, while Japanese is the most challenging target language, and (3) game genre (RPG vs. strategy) has minimal impact on accuracy. While closed-source models achieve the highest overall performance, open-weight alternatives such as Qwen-2.5-72B provide competitive quality at substantially lower cost. These findings provide practical guidance for deploying LLM-based LQA systems in production game localization workflows.
%U https://aclanthology.org/2026.amta-research.12/
%P 186-201
Markdown (Informal)
[Benchmarking Large Language Models for Game Localization Quality Assurance: A Cross-Model, Cross-Lingual Analysis](https://aclanthology.org/2026.amta-research.12/) (Tian & Wu, AMTA 2026)
ACL