@inproceedings{manaseryan-kennington-2026-adaptive,
title = "Adaptive Emotion Management in Human-robot Dialogue using Online Group Relative Policy Optimization",
author = "Manaseryan, Anna and
Kennington, Casey",
editor = "Choi, Jinho D. and
Chen, Yun-Nung and
Funakoshi, Kotaro and
Emami, Ali",
booktitle = "Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue",
month = aug,
year = "2026",
address = "Atlanta, Georgia, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.sigdial-1.41/",
pages = "585--596",
abstract = "Emotion expression is essential for human-robot interaction, yet current systems rely on static models that cannot adapt to individual users. We present an online reinforcement learning framework that adapts robot emotional behavior policy during live dialogue using binary human feedback. The system integrates a DeBERTa-v3-base emotion classifier and applies Group Relative Policy Optimization (GRPO) in a human-robot dialogue system. At each dialogue turn, the classifier samples a group of emotion candidates and the selected emotion is passed to a generative model that synthesizes a novel robot emotional behavior. We evaluate the system in three experiments: (1) offline supervised fine-tuning followed by GRPO on synthetic dialogue data, (2) a live GRPO training with a human teacher and (3) a final experiment with human participants. Results indicate that the robot was perceived as responsive and emotionally consistent, with high ratings for personality coherence and contextual appropriateness of emotional behaviors. Results further show that online GRPO with human feedback enables effective real-time emotion adaptation in embodied interaction."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="manaseryan-kennington-2026-adaptive">
<titleInfo>
<title>Adaptive Emotion Management in Human-robot Dialogue using Online Group Relative Policy Optimization</title>
</titleInfo>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Manaseryan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Casey</namePart>
<namePart type="family">Kennington</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jinho</namePart>
<namePart type="given">D</namePart>
<namePart type="family">Choi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yun-Nung</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kotaro</namePart>
<namePart type="family">Funakoshi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ali</namePart>
<namePart type="family">Emami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Atlanta, Georgia, USA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Emotion expression is essential for human-robot interaction, yet current systems rely on static models that cannot adapt to individual users. We present an online reinforcement learning framework that adapts robot emotional behavior policy during live dialogue using binary human feedback. The system integrates a DeBERTa-v3-base emotion classifier and applies Group Relative Policy Optimization (GRPO) in a human-robot dialogue system. At each dialogue turn, the classifier samples a group of emotion candidates and the selected emotion is passed to a generative model that synthesizes a novel robot emotional behavior. We evaluate the system in three experiments: (1) offline supervised fine-tuning followed by GRPO on synthetic dialogue data, (2) a live GRPO training with a human teacher and (3) a final experiment with human participants. Results indicate that the robot was perceived as responsive and emotionally consistent, with high ratings for personality coherence and contextual appropriateness of emotional behaviors. Results further show that online GRPO with human feedback enables effective real-time emotion adaptation in embodied interaction.</abstract>
<identifier type="citekey">manaseryan-kennington-2026-adaptive</identifier>
<location>
<url>https://aclanthology.org/2026.sigdial-1.41/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>585</start>
<end>596</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Adaptive Emotion Management in Human-robot Dialogue using Online Group Relative Policy Optimization
%A Manaseryan, Anna
%A Kennington, Casey
%Y Choi, Jinho D.
%Y Chen, Yun-Nung
%Y Funakoshi, Kotaro
%Y Emami, Ali
%S Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue
%D 2026
%8 August
%I Association for Computational Linguistics
%C Atlanta, Georgia, USA
%F manaseryan-kennington-2026-adaptive
%X Emotion expression is essential for human-robot interaction, yet current systems rely on static models that cannot adapt to individual users. We present an online reinforcement learning framework that adapts robot emotional behavior policy during live dialogue using binary human feedback. The system integrates a DeBERTa-v3-base emotion classifier and applies Group Relative Policy Optimization (GRPO) in a human-robot dialogue system. At each dialogue turn, the classifier samples a group of emotion candidates and the selected emotion is passed to a generative model that synthesizes a novel robot emotional behavior. We evaluate the system in three experiments: (1) offline supervised fine-tuning followed by GRPO on synthetic dialogue data, (2) a live GRPO training with a human teacher and (3) a final experiment with human participants. Results indicate that the robot was perceived as responsive and emotionally consistent, with high ratings for personality coherence and contextual appropriateness of emotional behaviors. Results further show that online GRPO with human feedback enables effective real-time emotion adaptation in embodied interaction.
%U https://aclanthology.org/2026.sigdial-1.41/
%P 585-596
Markdown (Informal)
[Adaptive Emotion Management in Human-robot Dialogue using Online Group Relative Policy Optimization](https://aclanthology.org/2026.sigdial-1.41/) (Manaseryan & Kennington, SIGDIAL 2026)
ACL