@inproceedings{masaki-kato-haris-2026-ping,
title = "Ping-Ponder: Concurrent Dual-Agent Spoken Dialogue via Shared Belief State",
author = "Masaki-Kato, Akiko and
Haris, Gulzar",
editor = "Choi, Jinho D. and
Chen, Yun-Nung and
Funakoshi, Kotaro and
Emami, Ali",
booktitle = "Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue",
month = aug,
year = "2026",
address = "Atlanta, Georgia, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.sigdial-1.14/",
pages = "188--204",
abstract = "Maintaining responsiveness while performing complex task reasoning remains a fundamental challenge in spoken dialogue systems. Existing dual-agent approaches separate conversation from reasoning but typically run the two sequentially, beginning reasoning only after the necessary user information is collected, causing substantial planning latency. We propose Ping-Ponder, a dual-agent framework in which two agents operate concurrently through a shared belief state. The conversational agent (Ping) handles real-time spoken interaction while incrementally extracting intent from user utterances and updating the shared belief state. The reasoning agent (Ponder) begins planning and knowledge retrieval as soon as partial intent information becomes available, without waiting for complete slot filling, and incrementally updates the belief state with intermediate results. This allows Ping to continue eliciting missing information while promptly presenting candidate plans generated by Ponder. In a paired single-operator case study, the proposed approach reduces average planning-phase latency by 93{\%} in English (from 7.1s to 0.5s) and 87{\%} in Japanese (from 7.3s to 1.0s) compared with a sequential dual-agent baseline. In a larger replay study on two spoken task-oriented corpora, the same ordering holds under controlled conditions: Ping-Ponder maintains time-to-first-response parity with the blocking baseline while reducing planning-phase latency on planning-triggered turns by 31{--}32{\%}, with no evidence of degraded reference-based response quality under an LLM judge; human-perceived conversational quality remains to be tested in a user study. These results suggest that concurrent processing through a shared belief state is a promising direction for reconciling responsiveness and reasoning capability in spoken dialogue systems."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="masaki-kato-haris-2026-ping">
<titleInfo>
<title>Ping-Ponder: Concurrent Dual-Agent Spoken Dialogue via Shared Belief State</title>
</titleInfo>
<name type="personal">
<namePart type="given">Akiko</namePart>
<namePart type="family">Masaki-Kato</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gulzar</namePart>
<namePart type="family">Haris</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jinho</namePart>
<namePart type="given">D</namePart>
<namePart type="family">Choi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yun-Nung</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kotaro</namePart>
<namePart type="family">Funakoshi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ali</namePart>
<namePart type="family">Emami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Atlanta, Georgia, USA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Maintaining responsiveness while performing complex task reasoning remains a fundamental challenge in spoken dialogue systems. Existing dual-agent approaches separate conversation from reasoning but typically run the two sequentially, beginning reasoning only after the necessary user information is collected, causing substantial planning latency. We propose Ping-Ponder, a dual-agent framework in which two agents operate concurrently through a shared belief state. The conversational agent (Ping) handles real-time spoken interaction while incrementally extracting intent from user utterances and updating the shared belief state. The reasoning agent (Ponder) begins planning and knowledge retrieval as soon as partial intent information becomes available, without waiting for complete slot filling, and incrementally updates the belief state with intermediate results. This allows Ping to continue eliciting missing information while promptly presenting candidate plans generated by Ponder. In a paired single-operator case study, the proposed approach reduces average planning-phase latency by 93% in English (from 7.1s to 0.5s) and 87% in Japanese (from 7.3s to 1.0s) compared with a sequential dual-agent baseline. In a larger replay study on two spoken task-oriented corpora, the same ordering holds under controlled conditions: Ping-Ponder maintains time-to-first-response parity with the blocking baseline while reducing planning-phase latency on planning-triggered turns by 31–32%, with no evidence of degraded reference-based response quality under an LLM judge; human-perceived conversational quality remains to be tested in a user study. These results suggest that concurrent processing through a shared belief state is a promising direction for reconciling responsiveness and reasoning capability in spoken dialogue systems.</abstract>
<identifier type="citekey">masaki-kato-haris-2026-ping</identifier>
<location>
<url>https://aclanthology.org/2026.sigdial-1.14/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>188</start>
<end>204</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Ping-Ponder: Concurrent Dual-Agent Spoken Dialogue via Shared Belief State
%A Masaki-Kato, Akiko
%A Haris, Gulzar
%Y Choi, Jinho D.
%Y Chen, Yun-Nung
%Y Funakoshi, Kotaro
%Y Emami, Ali
%S Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue
%D 2026
%8 August
%I Association for Computational Linguistics
%C Atlanta, Georgia, USA
%F masaki-kato-haris-2026-ping
%X Maintaining responsiveness while performing complex task reasoning remains a fundamental challenge in spoken dialogue systems. Existing dual-agent approaches separate conversation from reasoning but typically run the two sequentially, beginning reasoning only after the necessary user information is collected, causing substantial planning latency. We propose Ping-Ponder, a dual-agent framework in which two agents operate concurrently through a shared belief state. The conversational agent (Ping) handles real-time spoken interaction while incrementally extracting intent from user utterances and updating the shared belief state. The reasoning agent (Ponder) begins planning and knowledge retrieval as soon as partial intent information becomes available, without waiting for complete slot filling, and incrementally updates the belief state with intermediate results. This allows Ping to continue eliciting missing information while promptly presenting candidate plans generated by Ponder. In a paired single-operator case study, the proposed approach reduces average planning-phase latency by 93% in English (from 7.1s to 0.5s) and 87% in Japanese (from 7.3s to 1.0s) compared with a sequential dual-agent baseline. In a larger replay study on two spoken task-oriented corpora, the same ordering holds under controlled conditions: Ping-Ponder maintains time-to-first-response parity with the blocking baseline while reducing planning-phase latency on planning-triggered turns by 31–32%, with no evidence of degraded reference-based response quality under an LLM judge; human-perceived conversational quality remains to be tested in a user study. These results suggest that concurrent processing through a shared belief state is a promising direction for reconciling responsiveness and reasoning capability in spoken dialogue systems.
%U https://aclanthology.org/2026.sigdial-1.14/
%P 188-204
Markdown (Informal)
[Ping-Ponder: Concurrent Dual-Agent Spoken Dialogue via Shared Belief State](https://aclanthology.org/2026.sigdial-1.14/) (Masaki-Kato & Haris, SIGDIAL 2026)
ACL