@inproceedings{mao-2026-corpus,
title = "A Corpus-Based Profiling of Regional {E}nglish Variants in Global Media: Insights from Olympic Journalism",
author = "Mao, Felix",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.503/",
doi = "10.63317/3kzf8hoic5ht",
pages = "6340--6348",
abstract = "This paper investigates the distinctive linguistic characteristics of regional English variants through a quantitative analysis of global media coverage. The study applies advanced classification techniques, integrating GPT-based embeddings with Support Vector Machines, to a novel corpus, the Olympic Journalism English Variants Corpus. Comprising news articles related to Olympic Games covered by prominent news outlets in the United States, China, Spain, and Mexico between 2020 and 2023, this corpus enables a fine-grained analysis of 164 linguistic features across lexical, syntactic, readability, and sentiment dimensions. The findings reveal strong and interpretable distinctions in features such as verb ratio, nominality, and readability. This study not only demonstrated the enhanced classification capabilities of the model (optimized F1 score = 97.2), but also yielded deeper, data-driven stylistic analysis and insights of each English variant. This work provides a potential template that can be expanded to other World Englishes research."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mao-2026-corpus">
<titleInfo>
<title>A Corpus-Based Profiling of Regional English Variants in Global Media: Insights from Olympic Journalism</title>
</titleInfo>
<name type="personal">
<namePart type="given">Felix</namePart>
<namePart type="family">Mao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper investigates the distinctive linguistic characteristics of regional English variants through a quantitative analysis of global media coverage. The study applies advanced classification techniques, integrating GPT-based embeddings with Support Vector Machines, to a novel corpus, the Olympic Journalism English Variants Corpus. Comprising news articles related to Olympic Games covered by prominent news outlets in the United States, China, Spain, and Mexico between 2020 and 2023, this corpus enables a fine-grained analysis of 164 linguistic features across lexical, syntactic, readability, and sentiment dimensions. The findings reveal strong and interpretable distinctions in features such as verb ratio, nominality, and readability. This study not only demonstrated the enhanced classification capabilities of the model (optimized F1 score = 97.2), but also yielded deeper, data-driven stylistic analysis and insights of each English variant. This work provides a potential template that can be expanded to other World Englishes research.</abstract>
<identifier type="citekey">mao-2026-corpus</identifier>
<identifier type="doi">10.63317/3kzf8hoic5ht</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.503/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>6340</start>
<end>6348</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Corpus-Based Profiling of Regional English Variants in Global Media: Insights from Olympic Journalism
%A Mao, Felix
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F mao-2026-corpus
%X This paper investigates the distinctive linguistic characteristics of regional English variants through a quantitative analysis of global media coverage. The study applies advanced classification techniques, integrating GPT-based embeddings with Support Vector Machines, to a novel corpus, the Olympic Journalism English Variants Corpus. Comprising news articles related to Olympic Games covered by prominent news outlets in the United States, China, Spain, and Mexico between 2020 and 2023, this corpus enables a fine-grained analysis of 164 linguistic features across lexical, syntactic, readability, and sentiment dimensions. The findings reveal strong and interpretable distinctions in features such as verb ratio, nominality, and readability. This study not only demonstrated the enhanced classification capabilities of the model (optimized F1 score = 97.2), but also yielded deeper, data-driven stylistic analysis and insights of each English variant. This work provides a potential template that can be expanded to other World Englishes research.
%R 10.63317/3kzf8hoic5ht
%U https://aclanthology.org/2026.lrec-1.503/
%U https://doi.org/10.63317/3kzf8hoic5ht
%P 6340-6348
Markdown (Informal)
[A Corpus-Based Profiling of Regional English Variants in Global Media: Insights from Olympic Journalism](https://aclanthology.org/2026.lrec-1.503/) (Mao, LREC 2026)
ACL