@inproceedings{hosseinpourkhoshkbari-etal-2026-identifying,
title = "Identifying Communication Profiles from Physician{--}Patient Conversations Using Large Language Models",
author = "Hosseinpourkhoshkbari, Reyhaneh and
Jain, Yash Bipin and
Golden, Richard",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.31/",
pages = "235--247",
ISBN = "979-8-9983004-1-7",
abstract = "Large language models (LLMs) are increasingly used to score clinical communication, but validation is often limited to individual checklist items or total scores. These measures may not capture how communication behaviors occur together within a transcript. We therefore examine \textit{communication profiles}: recurring combinations of behaviors that characterize different patterns of clinician communication and may support more targeted formative feedback. We analyzed 213 simulated respiratory OSCE transcripts rated on 15 binary Kalamazoo-derived communication items and compared final human ratings with GPT-4o, GPT-o3, and GPT-5.6 Sol. Raw item-level agreement was relatively high overall, but varied substantially across behaviors and was lower for several judgment-intensive items used for profile modeling. Bayesian latent-class analysis identified three stable human-derived profiles. Although the LLM-derived profiles showed broadly similar item-probability patterns, the models frequently assigned individual transcripts to different profiles than the human ratings. These findings show that agreement on individual communication skills does not necessarily translate into agreement on higher-level communication profiles. If LLMs are used to provide profile-based feedback, validation should therefore include profile-level agreement in addition to item-level performance."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hosseinpourkhoshkbari-etal-2026-identifying">
<titleInfo>
<title>Identifying Communication Profiles from Physician–Patient Conversations Using Large Language Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Reyhaneh</namePart>
<namePart type="family">Hosseinpourkhoshkbari</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yash</namePart>
<namePart type="given">Bipin</namePart>
<namePart type="family">Jain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Richard</namePart>
<namePart type="family">Golden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Large language models (LLMs) are increasingly used to score clinical communication, but validation is often limited to individual checklist items or total scores. These measures may not capture how communication behaviors occur together within a transcript. We therefore examine communication profiles: recurring combinations of behaviors that characterize different patterns of clinician communication and may support more targeted formative feedback. We analyzed 213 simulated respiratory OSCE transcripts rated on 15 binary Kalamazoo-derived communication items and compared final human ratings with GPT-4o, GPT-o3, and GPT-5.6 Sol. Raw item-level agreement was relatively high overall, but varied substantially across behaviors and was lower for several judgment-intensive items used for profile modeling. Bayesian latent-class analysis identified three stable human-derived profiles. Although the LLM-derived profiles showed broadly similar item-probability patterns, the models frequently assigned individual transcripts to different profiles than the human ratings. These findings show that agreement on individual communication skills does not necessarily translate into agreement on higher-level communication profiles. If LLMs are used to provide profile-based feedback, validation should therefore include profile-level agreement in addition to item-level performance.</abstract>
<identifier type="citekey">hosseinpourkhoshkbari-etal-2026-identifying</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.31/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>235</start>
<end>247</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Identifying Communication Profiles from Physician–Patient Conversations Using Large Language Models
%A Hosseinpourkhoshkbari, Reyhaneh
%A Jain, Yash Bipin
%A Golden, Richard
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F hosseinpourkhoshkbari-etal-2026-identifying
%X Large language models (LLMs) are increasingly used to score clinical communication, but validation is often limited to individual checklist items or total scores. These measures may not capture how communication behaviors occur together within a transcript. We therefore examine communication profiles: recurring combinations of behaviors that characterize different patterns of clinician communication and may support more targeted formative feedback. We analyzed 213 simulated respiratory OSCE transcripts rated on 15 binary Kalamazoo-derived communication items and compared final human ratings with GPT-4o, GPT-o3, and GPT-5.6 Sol. Raw item-level agreement was relatively high overall, but varied substantially across behaviors and was lower for several judgment-intensive items used for profile modeling. Bayesian latent-class analysis identified three stable human-derived profiles. Although the LLM-derived profiles showed broadly similar item-probability patterns, the models frequently assigned individual transcripts to different profiles than the human ratings. These findings show that agreement on individual communication skills does not necessarily translate into agreement on higher-level communication profiles. If LLMs are used to provide profile-based feedback, validation should therefore include profile-level agreement in addition to item-level performance.
%U https://aclanthology.org/2026.aimecon-wip.31/
%P 235-247
Markdown (Informal)
[Identifying Communication Profiles from Physician–Patient Conversations Using Large Language Models](https://aclanthology.org/2026.aimecon-wip.31/) (Hosseinpourkhoshkbari et al., AIME-Con 2026)
ACL