@inproceedings{tran-2026-tt501,
title = "tt501 at {A}rch{EHR}-{QA} 2026: Few-Shot Prompting with Retrieval-Augmented Generation for Grounded Clinical {EHR} Question Answering",
author = "Tran, Tai Tan",
editor = "Gupta, Deepak and
Thompson, Paul and
Ananiadou, Sophia and
Demner-Fushman, Dina",
booktitle = "Proceedings of the Third Workshop on Patient-Oriented Language Processing ({CL}4{H}ealth) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cl4health-1.46/",
doi = "10.63317/2tjqwa7c7nqf",
pages = "497--505",
abstract = "We present the ArchEHR-QA 2026 shared task system of team tt501, which addresses evidence identification (Subtask 2), answer generation (Subtask 3), and evidence alignment (Subtask 4) from electronic health record notes. Our approach relies entirely on prompt engineering with xAI{'}s Grok models, without any task-specific fine-tuning or external knowledge. For evidence identification we compare a hybrid BM25 plus large language model (LLM) reranker with a full-context chain-of-thought ensemble and refinement step, finding that full-note reasoning yields higher recall and F1. For answer generation we implement a retrieval-augmented generation pipeline that conditions on predicted evidence sentences and few-shot examples, improving lexical and semantic faithfulness over a zero-shot baseline. For evidence alignment we design a recall-oriented few-shot prompt enriched with explicit rationales that teach the model how to map each answer sentence back to its supporting note sentences. We report official shared task results and analyse the impact of these design choices across the three subtasks."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tran-2026-tt501">
<titleInfo>
<title>tt501 at ArchEHR-QA 2026: Few-Shot Prompting with Retrieval-Augmented Generation for Grounded Clinical EHR Question Answering</title>
</titleInfo>
<name type="personal">
<namePart type="given">Tai</namePart>
<namePart type="given">Tan</namePart>
<namePart type="family">Tran</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on Patient-Oriented Language Processing (CL4Health) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Deepak</namePart>
<namePart type="family">Gupta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Thompson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sophia</namePart>
<namePart type="family">Ananiadou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dina</namePart>
<namePart type="family">Demner-Fushman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present the ArchEHR-QA 2026 shared task system of team tt501, which addresses evidence identification (Subtask 2), answer generation (Subtask 3), and evidence alignment (Subtask 4) from electronic health record notes. Our approach relies entirely on prompt engineering with xAI’s Grok models, without any task-specific fine-tuning or external knowledge. For evidence identification we compare a hybrid BM25 plus large language model (LLM) reranker with a full-context chain-of-thought ensemble and refinement step, finding that full-note reasoning yields higher recall and F1. For answer generation we implement a retrieval-augmented generation pipeline that conditions on predicted evidence sentences and few-shot examples, improving lexical and semantic faithfulness over a zero-shot baseline. For evidence alignment we design a recall-oriented few-shot prompt enriched with explicit rationales that teach the model how to map each answer sentence back to its supporting note sentences. We report official shared task results and analyse the impact of these design choices across the three subtasks.</abstract>
<identifier type="citekey">tran-2026-tt501</identifier>
<identifier type="doi">10.63317/2tjqwa7c7nqf</identifier>
<location>
<url>https://aclanthology.org/2026.cl4health-1.46/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>497</start>
<end>505</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T tt501 at ArchEHR-QA 2026: Few-Shot Prompting with Retrieval-Augmented Generation for Grounded Clinical EHR Question Answering
%A Tran, Tai Tan
%Y Gupta, Deepak
%Y Thompson, Paul
%Y Ananiadou, Sophia
%Y Demner-Fushman, Dina
%S Proceedings of the Third Workshop on Patient-Oriented Language Processing (CL4Health) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F tran-2026-tt501
%X We present the ArchEHR-QA 2026 shared task system of team tt501, which addresses evidence identification (Subtask 2), answer generation (Subtask 3), and evidence alignment (Subtask 4) from electronic health record notes. Our approach relies entirely on prompt engineering with xAI’s Grok models, without any task-specific fine-tuning or external knowledge. For evidence identification we compare a hybrid BM25 plus large language model (LLM) reranker with a full-context chain-of-thought ensemble and refinement step, finding that full-note reasoning yields higher recall and F1. For answer generation we implement a retrieval-augmented generation pipeline that conditions on predicted evidence sentences and few-shot examples, improving lexical and semantic faithfulness over a zero-shot baseline. For evidence alignment we design a recall-oriented few-shot prompt enriched with explicit rationales that teach the model how to map each answer sentence back to its supporting note sentences. We report official shared task results and analyse the impact of these design choices across the three subtasks.
%R 10.63317/2tjqwa7c7nqf
%U https://aclanthology.org/2026.cl4health-1.46/
%U https://doi.org/10.63317/2tjqwa7c7nqf
%P 497-505
Markdown (Informal)
[tt501 at ArchEHR-QA 2026: Few-Shot Prompting with Retrieval-Augmented Generation for Grounded Clinical EHR Question Answering](https://aclanthology.org/2026.cl4health-1.46/) (Tran, CL4Health 2026)
ACL