@inproceedings{ng-etal-2026-coach,
title = "{COACH} Meets {QUORUM}: A Framework and Pipeline for Aligning User, Expert, and Developer Perspectives in {LLM}-Generated Health Counselling",
author = "Ng, Yee Man and
van Dijk, Bram and
Beynen, Pieter and
Boekesteijn, Otto and
Jansen, Joris and
van Oortmerssen, Gerard and
van Duijn, Max J. and
Spruit, Marco",
editor = "Gupta, Deepak and
Thompson, Paul and
Ananiadou, Sophia and
Demner-Fushman, Dina",
booktitle = "Proceedings of the Third Workshop on Patient-Oriented Language Processing ({CL}4{H}ealth) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cl4health-1.2/",
doi = "10.63317/4x7koa8w32ny",
pages = "15--25",
abstract = "Systems that collect data on sleep, mood, and activities can provide valuable lifestyle counselling to populations affected by chronic disease and its consequences. Such systems are, however, challenging to develop; in addition to reliably extracting patterns from user-specific data, systems should contextualise these patterns with validated medical knowledge to ensure the quality of counselling and generate counselling that is relevant to a real user. We present QUORUM, an evaluation framework that unifies these developer-, expert-, and user-centric perspectives, and show with a real case study that it meaningfully tracks convergence and divergence in stakeholder perspectives. We also present COACH, a Large Language Model-driven pipeline to generate personalised lifestyle counselling for our Healthy Chronos use case, a diary app for cancer patients and survivors. Applying our framework indicates that, overall, users, medical experts, and developers converge on the view that the generated counselling is relevant, of good quality, and reliable. However, stakeholders also diverge on the tone of the counselling, sensitivity to errors in pattern-extraction, and potential hallucinations. These findings highlight the importance of multi-stakeholder evaluation for consumer health language technologies and illustrate how a unified evaluation framework can support trustworthy, patient-centered NLP systems in real-world settings."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ng-etal-2026-coach">
<titleInfo>
<title>COACH Meets QUORUM: A Framework and Pipeline for Aligning User, Expert, and Developer Perspectives in LLM-Generated Health Counselling</title>
</titleInfo>
<name type="personal">
<namePart type="given">Yee</namePart>
<namePart type="given">Man</namePart>
<namePart type="family">Ng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bram</namePart>
<namePart type="family">van Dijk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pieter</namePart>
<namePart type="family">Beynen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Otto</namePart>
<namePart type="family">Boekesteijn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joris</namePart>
<namePart type="family">Jansen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gerard</namePart>
<namePart type="family">van Oortmerssen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Max</namePart>
<namePart type="given">J</namePart>
<namePart type="family">van Duijn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="family">Spruit</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on Patient-Oriented Language Processing (CL4Health) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Deepak</namePart>
<namePart type="family">Gupta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Thompson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sophia</namePart>
<namePart type="family">Ananiadou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dina</namePart>
<namePart type="family">Demner-Fushman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Systems that collect data on sleep, mood, and activities can provide valuable lifestyle counselling to populations affected by chronic disease and its consequences. Such systems are, however, challenging to develop; in addition to reliably extracting patterns from user-specific data, systems should contextualise these patterns with validated medical knowledge to ensure the quality of counselling and generate counselling that is relevant to a real user. We present QUORUM, an evaluation framework that unifies these developer-, expert-, and user-centric perspectives, and show with a real case study that it meaningfully tracks convergence and divergence in stakeholder perspectives. We also present COACH, a Large Language Model-driven pipeline to generate personalised lifestyle counselling for our Healthy Chronos use case, a diary app for cancer patients and survivors. Applying our framework indicates that, overall, users, medical experts, and developers converge on the view that the generated counselling is relevant, of good quality, and reliable. However, stakeholders also diverge on the tone of the counselling, sensitivity to errors in pattern-extraction, and potential hallucinations. These findings highlight the importance of multi-stakeholder evaluation for consumer health language technologies and illustrate how a unified evaluation framework can support trustworthy, patient-centered NLP systems in real-world settings.</abstract>
<identifier type="citekey">ng-etal-2026-coach</identifier>
<identifier type="doi">10.63317/4x7koa8w32ny</identifier>
<location>
<url>https://aclanthology.org/2026.cl4health-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>15</start>
<end>25</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T COACH Meets QUORUM: A Framework and Pipeline for Aligning User, Expert, and Developer Perspectives in LLM-Generated Health Counselling
%A Ng, Yee Man
%A van Dijk, Bram
%A Beynen, Pieter
%A Boekesteijn, Otto
%A Jansen, Joris
%A van Oortmerssen, Gerard
%A van Duijn, Max J.
%A Spruit, Marco
%Y Gupta, Deepak
%Y Thompson, Paul
%Y Ananiadou, Sophia
%Y Demner-Fushman, Dina
%S Proceedings of the Third Workshop on Patient-Oriented Language Processing (CL4Health) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F ng-etal-2026-coach
%X Systems that collect data on sleep, mood, and activities can provide valuable lifestyle counselling to populations affected by chronic disease and its consequences. Such systems are, however, challenging to develop; in addition to reliably extracting patterns from user-specific data, systems should contextualise these patterns with validated medical knowledge to ensure the quality of counselling and generate counselling that is relevant to a real user. We present QUORUM, an evaluation framework that unifies these developer-, expert-, and user-centric perspectives, and show with a real case study that it meaningfully tracks convergence and divergence in stakeholder perspectives. We also present COACH, a Large Language Model-driven pipeline to generate personalised lifestyle counselling for our Healthy Chronos use case, a diary app for cancer patients and survivors. Applying our framework indicates that, overall, users, medical experts, and developers converge on the view that the generated counselling is relevant, of good quality, and reliable. However, stakeholders also diverge on the tone of the counselling, sensitivity to errors in pattern-extraction, and potential hallucinations. These findings highlight the importance of multi-stakeholder evaluation for consumer health language technologies and illustrate how a unified evaluation framework can support trustworthy, patient-centered NLP systems in real-world settings.
%R 10.63317/4x7koa8w32ny
%U https://aclanthology.org/2026.cl4health-1.2/
%U https://doi.org/10.63317/4x7koa8w32ny
%P 15-25
Markdown (Informal)
[COACH Meets QUORUM: A Framework and Pipeline for Aligning User, Expert, and Developer Perspectives in LLM-Generated Health Counselling](https://aclanthology.org/2026.cl4health-1.2/) (Ng et al., CL4Health 2026)
ACL
- Yee Man Ng, Bram van Dijk, Pieter Beynen, Otto Boekesteijn, Joris Jansen, Gerard van Oortmerssen, Max J. van Duijn, and Marco Spruit. 2026. COACH Meets QUORUM: A Framework and Pipeline for Aligning User, Expert, and Developer Perspectives in LLM-Generated Health Counselling. In Proceedings of the Third Workshop on Patient-Oriented Language Processing (CL4Health) @ LREC 2026, pages 15–25, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).