@inproceedings{milano-etal-2026-language,
title = "Language Ideologies in a Multilingual Society: An {LLM}-based Analysis of {L}uxembourgish News Comments",
author = "Milano, Emilia and
Plum, Alistair and
Scherrer, Yves and
Purschke, Christoph",
editor = "Stranisci, Marco Antonio and
Falk, Neele and
Labat, Sofie and
Lo, Soda Marem and
Velutharambath, Aswathy and
Weber, Sabine and
Damiano, Rossana and
Frenda, Simona and
Hoste, Veronique and
Kleinberg, Bennett and
Klinger, Roman and
Patti, Viviana and
Plaza-del-Arco, Flor Miriam and
Sap, Maarten and
Yimam, Seid Muhie",
booktitle = "Proceedings of the 1st Workshop on Social Context ({S}o{C}on) and the 2nd Workshop on Integrating {NLP} and Psychology to Study Social Interactions ({NLPSI}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "European Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.socon-1.12/",
doi = "10.63317/48kwvvr63gms",
pages = "114--131",
abstract = "Detecting language ideologies is a valuable yet complex task for understanding how identities are constructed through discourse. In Luxembourg{'}s multicultural and multilingual society, language ideologies reflect more than simple preferences: they carry deep cultural and social meanings, shaping identities and social belonging. Following recent developments in applying Natural Language Processing tools to linguistics and social science, this paper explores the potential of large language models to assist in the detection of language ideologies. We manually annotate a corpus of user comments in Luxembourgish with predefined ideological categories and then evaluate the performance of large language models under varying prompt conditions to assess their ability to replicate these human annotations. Since Luxembourgish is a small language and poorly represented in the LLMs' training data, we also investigate whether machine-translating the data to high-resource languages increases performance on the ideology detection task. Our findings suggest that, while LLMs are not yet fully optimized for a multi-class ideological annotation task, they are practical tools to identify language ideological content."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="milano-etal-2026-language">
<titleInfo>
<title>Language Ideologies in a Multilingual Society: An LLM-based Analysis of Luxembourgish News Comments</title>
</titleInfo>
<name type="personal">
<namePart type="given">Emilia</namePart>
<namePart type="family">Milano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alistair</namePart>
<namePart type="family">Plum</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yves</namePart>
<namePart type="family">Scherrer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christoph</namePart>
<namePart type="family">Purschke</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 1st Workshop on Social Context (SoCon) and the 2nd Workshop on Integrating NLP and Psychology to Study Social Interactions (NLPSI) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="given">Antonio</namePart>
<namePart type="family">Stranisci</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Neele</namePart>
<namePart type="family">Falk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sofie</namePart>
<namePart type="family">Labat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Soda</namePart>
<namePart type="given">Marem</namePart>
<namePart type="family">Lo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aswathy</namePart>
<namePart type="family">Velutharambath</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sabine</namePart>
<namePart type="family">Weber</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rossana</namePart>
<namePart type="family">Damiano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simona</namePart>
<namePart type="family">Frenda</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Veronique</namePart>
<namePart type="family">Hoste</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bennett</namePart>
<namePart type="family">Kleinberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Roman</namePart>
<namePart type="family">Klinger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviana</namePart>
<namePart type="family">Patti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Flor</namePart>
<namePart type="given">Miriam</namePart>
<namePart type="family">Plaza-del-Arco</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maarten</namePart>
<namePart type="family">Sap</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Seid</namePart>
<namePart type="given">Muhie</namePart>
<namePart type="family">Yimam</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Detecting language ideologies is a valuable yet complex task for understanding how identities are constructed through discourse. In Luxembourg’s multicultural and multilingual society, language ideologies reflect more than simple preferences: they carry deep cultural and social meanings, shaping identities and social belonging. Following recent developments in applying Natural Language Processing tools to linguistics and social science, this paper explores the potential of large language models to assist in the detection of language ideologies. We manually annotate a corpus of user comments in Luxembourgish with predefined ideological categories and then evaluate the performance of large language models under varying prompt conditions to assess their ability to replicate these human annotations. Since Luxembourgish is a small language and poorly represented in the LLMs’ training data, we also investigate whether machine-translating the data to high-resource languages increases performance on the ideology detection task. Our findings suggest that, while LLMs are not yet fully optimized for a multi-class ideological annotation task, they are practical tools to identify language ideological content.</abstract>
<identifier type="citekey">milano-etal-2026-language</identifier>
<identifier type="doi">10.63317/48kwvvr63gms</identifier>
<location>
<url>https://aclanthology.org/2026.socon-1.12/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>114</start>
<end>131</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Language Ideologies in a Multilingual Society: An LLM-based Analysis of Luxembourgish News Comments
%A Milano, Emilia
%A Plum, Alistair
%A Scherrer, Yves
%A Purschke, Christoph
%Y Stranisci, Marco Antonio
%Y Falk, Neele
%Y Labat, Sofie
%Y Lo, Soda Marem
%Y Velutharambath, Aswathy
%Y Weber, Sabine
%Y Damiano, Rossana
%Y Frenda, Simona
%Y Hoste, Veronique
%Y Kleinberg, Bennett
%Y Klinger, Roman
%Y Patti, Viviana
%Y Plaza-del-Arco, Flor Miriam
%Y Sap, Maarten
%Y Yimam, Seid Muhie
%S Proceedings of the 1st Workshop on Social Context (SoCon) and the 2nd Workshop on Integrating NLP and Psychology to Study Social Interactions (NLPSI) @ LREC 2026
%D 2026
%8 May
%I European Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F milano-etal-2026-language
%X Detecting language ideologies is a valuable yet complex task for understanding how identities are constructed through discourse. In Luxembourg’s multicultural and multilingual society, language ideologies reflect more than simple preferences: they carry deep cultural and social meanings, shaping identities and social belonging. Following recent developments in applying Natural Language Processing tools to linguistics and social science, this paper explores the potential of large language models to assist in the detection of language ideologies. We manually annotate a corpus of user comments in Luxembourgish with predefined ideological categories and then evaluate the performance of large language models under varying prompt conditions to assess their ability to replicate these human annotations. Since Luxembourgish is a small language and poorly represented in the LLMs’ training data, we also investigate whether machine-translating the data to high-resource languages increases performance on the ideology detection task. Our findings suggest that, while LLMs are not yet fully optimized for a multi-class ideological annotation task, they are practical tools to identify language ideological content.
%R 10.63317/48kwvvr63gms
%U https://aclanthology.org/2026.socon-1.12/
%U https://doi.org/10.63317/48kwvvr63gms
%P 114-131
Markdown (Informal)
[Language Ideologies in a Multilingual Society: An LLM-based Analysis of Luxembourgish News Comments](https://aclanthology.org/2026.socon-1.12/) (Milano et al., SoCon-NLPSI 2026)
ACL