@inproceedings{kanishcheva-shvedova-2026-quantifying,
title = "Quantifying Code-Switching in a {U}krainian Parliamentary Dataset 1990-2021",
author = "Kanishcheva, Olha and
Shvedova, Maria",
editor = "Eskevich, Maria and
Vandeghinste, Vincent and
Bodron, David",
booktitle = "Proceedings of the {P}arla{CLARIN} {V} Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora",
month = may,
year = "2026",
address = "Palma de Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.parlaclarin-1.2/",
doi = "10.63317/42ss3zvt76a7",
pages = "2--12",
abstract = "Analyzing code-switching {--} the practice of mixing multiple languages in one discourse {--} remains a significant task in natural language processing (NLP). This study examines the Ukrainian-Russian bilingual context, focusing on quantifying language alternation in a multilingual dataset. We introduce metrics to assess linguistic boundaries and patterns, specifically addressing the complexities of processing texts where Ukrainian and Russian are used interchangeably, including word-level hybridization. Using a corpus of approximately 200,000 tokens derived from parliamentary transcripts (1990-2021), we apply code-switching metrics to identify frequency and patterns of language use. Our findings provide insights into bilingual communication dynamics and can be used to improve language identification models for mixed-language data."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kanishcheva-shvedova-2026-quantifying">
<titleInfo>
<title>Quantifying Code-Switching in a Ukrainian Parliamentary Dataset 1990-2021</title>
</titleInfo>
<name type="personal">
<namePart type="given">Olha</namePart>
<namePart type="family">Kanishcheva</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Shvedova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the ParlaCLARIN V Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Eskevich</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vincent</namePart>
<namePart type="family">Vandeghinste</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Bodron</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Analyzing code-switching – the practice of mixing multiple languages in one discourse – remains a significant task in natural language processing (NLP). This study examines the Ukrainian-Russian bilingual context, focusing on quantifying language alternation in a multilingual dataset. We introduce metrics to assess linguistic boundaries and patterns, specifically addressing the complexities of processing texts where Ukrainian and Russian are used interchangeably, including word-level hybridization. Using a corpus of approximately 200,000 tokens derived from parliamentary transcripts (1990-2021), we apply code-switching metrics to identify frequency and patterns of language use. Our findings provide insights into bilingual communication dynamics and can be used to improve language identification models for mixed-language data.</abstract>
<identifier type="citekey">kanishcheva-shvedova-2026-quantifying</identifier>
<identifier type="doi">10.63317/42ss3zvt76a7</identifier>
<location>
<url>https://aclanthology.org/2026.parlaclarin-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>2</start>
<end>12</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Quantifying Code-Switching in a Ukrainian Parliamentary Dataset 1990-2021
%A Kanishcheva, Olha
%A Shvedova, Maria
%Y Eskevich, Maria
%Y Vandeghinste, Vincent
%Y Bodron, David
%S Proceedings of the ParlaCLARIN V Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca (Spain)
%F kanishcheva-shvedova-2026-quantifying
%X Analyzing code-switching – the practice of mixing multiple languages in one discourse – remains a significant task in natural language processing (NLP). This study examines the Ukrainian-Russian bilingual context, focusing on quantifying language alternation in a multilingual dataset. We introduce metrics to assess linguistic boundaries and patterns, specifically addressing the complexities of processing texts where Ukrainian and Russian are used interchangeably, including word-level hybridization. Using a corpus of approximately 200,000 tokens derived from parliamentary transcripts (1990-2021), we apply code-switching metrics to identify frequency and patterns of language use. Our findings provide insights into bilingual communication dynamics and can be used to improve language identification models for mixed-language data.
%R 10.63317/42ss3zvt76a7
%U https://aclanthology.org/2026.parlaclarin-1.2/
%U https://doi.org/10.63317/42ss3zvt76a7
%P 2-12
Markdown (Informal)
[Quantifying Code-Switching in a Ukrainian Parliamentary Dataset 1990-2021](https://aclanthology.org/2026.parlaclarin-1.2/) (Kanishcheva & Shvedova, ParlaCLARIN 2026)
ACL