@inproceedings{tu-zahra-etal-2026-cross,
title = "Cross-Domain Evaluation of Transformer-Based Models for {P}unjabi Speech Emotion Recognition",
author = "Tu Zahra, Fatima and
Asim, Kulsoom and
Kumar, Sandesh and
Samad, Abdul",
editor = "Sarveswaran, Kengatharaiyer and
Vaidya, Ashwini",
booktitle = "Proceedings of the Second workshop on Challenges in Processing {S}outh {A}sian Languages ({CH}i{PSAL}2026)",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.chipsal-1.7/",
doi = "10.63317/3ao5m9p6ni9x",
pages = "59--67",
abstract = "Speech Emotion Recognition (SER) is an important part of human{--}computer interaction, but most existing research focuses on high-resource languages, with very limited work on regional languages such as Punjabi. This paper focuses on detecting emotions from Punjabi speech using machine learning and deep learning techniques. We curated our own Punjabi speech emotion dataset using volunteer recordings and real-world sources, covering four emotion classes: angry, happy, sad, and neutral. The data was preprocessed for consistency and evaluated using a multi-strategy framework (E1{--}E4) to test domain generalization. Three models were evaluated: CNN, ResNet-34, and the transformer-based Wav2Vec 2.0. Among these, the ResNet-34 model performed the best in the combined-domain strategy (E4), achieving a test accuracy of 96{\%}. While cross-corpus evaluations (E2, E3) highlighted challenges in generalizing to neutral emotions, the model achieved perfect scores for happy and sad classes in E4. These results demonstrate the effectiveness of residual networks and combined-domain training for emotion recognition in low-resource languages and highlight the potential for further work on Punjabi SER."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tu-zahra-etal-2026-cross">
<titleInfo>
<title>Cross-Domain Evaluation of Transformer-Based Models for Punjabi Speech Emotion Recognition</title>
</titleInfo>
<name type="personal">
<namePart type="given">Fatima</namePart>
<namePart type="family">Tu Zahra</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kulsoom</namePart>
<namePart type="family">Asim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sandesh</namePart>
<namePart type="family">Kumar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abdul</namePart>
<namePart type="family">Samad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second workshop on Challenges in Processing South Asian Languages (CHiPSAL2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kengatharaiyer</namePart>
<namePart type="family">Sarveswaran</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ashwini</namePart>
<namePart type="family">Vaidya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Speech Emotion Recognition (SER) is an important part of human–computer interaction, but most existing research focuses on high-resource languages, with very limited work on regional languages such as Punjabi. This paper focuses on detecting emotions from Punjabi speech using machine learning and deep learning techniques. We curated our own Punjabi speech emotion dataset using volunteer recordings and real-world sources, covering four emotion classes: angry, happy, sad, and neutral. The data was preprocessed for consistency and evaluated using a multi-strategy framework (E1–E4) to test domain generalization. Three models were evaluated: CNN, ResNet-34, and the transformer-based Wav2Vec 2.0. Among these, the ResNet-34 model performed the best in the combined-domain strategy (E4), achieving a test accuracy of 96%. While cross-corpus evaluations (E2, E3) highlighted challenges in generalizing to neutral emotions, the model achieved perfect scores for happy and sad classes in E4. These results demonstrate the effectiveness of residual networks and combined-domain training for emotion recognition in low-resource languages and highlight the potential for further work on Punjabi SER.</abstract>
<identifier type="citekey">tu-zahra-etal-2026-cross</identifier>
<identifier type="doi">10.63317/3ao5m9p6ni9x</identifier>
<location>
<url>https://aclanthology.org/2026.chipsal-1.7/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>59</start>
<end>67</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Cross-Domain Evaluation of Transformer-Based Models for Punjabi Speech Emotion Recognition
%A Tu Zahra, Fatima
%A Asim, Kulsoom
%A Kumar, Sandesh
%A Samad, Abdul
%Y Sarveswaran, Kengatharaiyer
%Y Vaidya, Ashwini
%S Proceedings of the Second workshop on Challenges in Processing South Asian Languages (CHiPSAL2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F tu-zahra-etal-2026-cross
%X Speech Emotion Recognition (SER) is an important part of human–computer interaction, but most existing research focuses on high-resource languages, with very limited work on regional languages such as Punjabi. This paper focuses on detecting emotions from Punjabi speech using machine learning and deep learning techniques. We curated our own Punjabi speech emotion dataset using volunteer recordings and real-world sources, covering four emotion classes: angry, happy, sad, and neutral. The data was preprocessed for consistency and evaluated using a multi-strategy framework (E1–E4) to test domain generalization. Three models were evaluated: CNN, ResNet-34, and the transformer-based Wav2Vec 2.0. Among these, the ResNet-34 model performed the best in the combined-domain strategy (E4), achieving a test accuracy of 96%. While cross-corpus evaluations (E2, E3) highlighted challenges in generalizing to neutral emotions, the model achieved perfect scores for happy and sad classes in E4. These results demonstrate the effectiveness of residual networks and combined-domain training for emotion recognition in low-resource languages and highlight the potential for further work on Punjabi SER.
%R 10.63317/3ao5m9p6ni9x
%U https://aclanthology.org/2026.chipsal-1.7/
%U https://doi.org/10.63317/3ao5m9p6ni9x
%P 59-67
Markdown (Informal)
[Cross-Domain Evaluation of Transformer-Based Models for Punjabi Speech Emotion Recognition](https://aclanthology.org/2026.chipsal-1.7/) (Tu Zahra et al., CHiPSAL 2026)
ACL