@inproceedings{aryoyudanta-etal-2026-night,
title = "Night Shift Nerds at {MEDIQA}-{SYNUR} 2026: Pushing Small Large Language Model Capability for Clinical Observation Extraction and Normalization from Nurse Dictation using {RLVR}",
author = "Aryoyudanta, Bayu and
Yuliana, Maria and
Rachman, Mikie and
Setiawan, I Made Agus",
editor = "Ben Abacha, Asma and
Bethard, Steven and
Bitterman, Danielle and
Naumann, Tristan and
Roberts, Kirk",
booktitle = "Proceedings of the 8th Workshop on Clinical Natural Language Processing (Clinical {NLP}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.clinicalnlp-1.18/",
doi = "10.63317/328obtnwpmsw",
pages = "163--173",
abstract = "We presented a small decoder-only language model for clinical observation extraction and normalization from nurse dictation developed using Reinforcement Learning with Verifiable Rewards (RLVR). We fine-tune Qwen3-1.7B model using a two-stage pipeline: (1) supervised fine-tuning (SFT) with an augmented chain-of-thought (CoT) dataset generated by a teacher model to mitigate RL cold-start, followed by (2) GRPO-based RLVR with multi-component reward functions that verify output format, concept presence, value type, and value correctness using the shared-task ontology (193 concepts) as a verifier. On the development set, SFT+GRPO substantially outperforms GRPO-only (F1 0.803 vs. 0.620). After the test holdout was released, our final system achieved 0.700 precision, 0.785 recall, and 0.740 F1. Error analysis shows remaining challenges in concept over-detection and missed concepts, as well as boundary errors in categorical and multi-select value type extraction. Our results demonstrates that small language models can enable accurate, cost-effective, and privacy preserving automated clinical documentation for nurse dictation, supporting scalable deployment in low-resource healthcare settings to reduce nurses' documentation burden."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="aryoyudanta-etal-2026-night">
<titleInfo>
<title>Night Shift Nerds at MEDIQA-SYNUR 2026: Pushing Small Large Language Model Capability for Clinical Observation Extraction and Normalization from Nurse Dictation using RLVR</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bayu</namePart>
<namePart type="family">Aryoyudanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Yuliana</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikie</namePart>
<namePart type="family">Rachman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">I</namePart>
<namePart type="given">Made</namePart>
<namePart type="given">Agus</namePart>
<namePart type="family">Setiawan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 8th Workshop on Clinical Natural Language Processing (Clinical NLP) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Asma</namePart>
<namePart type="family">Ben Abacha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Steven</namePart>
<namePart type="family">Bethard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Danielle</namePart>
<namePart type="family">Bitterman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tristan</namePart>
<namePart type="family">Naumann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kirk</namePart>
<namePart type="family">Roberts</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We presented a small decoder-only language model for clinical observation extraction and normalization from nurse dictation developed using Reinforcement Learning with Verifiable Rewards (RLVR). We fine-tune Qwen3-1.7B model using a two-stage pipeline: (1) supervised fine-tuning (SFT) with an augmented chain-of-thought (CoT) dataset generated by a teacher model to mitigate RL cold-start, followed by (2) GRPO-based RLVR with multi-component reward functions that verify output format, concept presence, value type, and value correctness using the shared-task ontology (193 concepts) as a verifier. On the development set, SFT+GRPO substantially outperforms GRPO-only (F1 0.803 vs. 0.620). After the test holdout was released, our final system achieved 0.700 precision, 0.785 recall, and 0.740 F1. Error analysis shows remaining challenges in concept over-detection and missed concepts, as well as boundary errors in categorical and multi-select value type extraction. Our results demonstrates that small language models can enable accurate, cost-effective, and privacy preserving automated clinical documentation for nurse dictation, supporting scalable deployment in low-resource healthcare settings to reduce nurses’ documentation burden.</abstract>
<identifier type="citekey">aryoyudanta-etal-2026-night</identifier>
<identifier type="doi">10.63317/328obtnwpmsw</identifier>
<location>
<url>https://aclanthology.org/2026.clinicalnlp-1.18/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>163</start>
<end>173</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Night Shift Nerds at MEDIQA-SYNUR 2026: Pushing Small Large Language Model Capability for Clinical Observation Extraction and Normalization from Nurse Dictation using RLVR
%A Aryoyudanta, Bayu
%A Yuliana, Maria
%A Rachman, Mikie
%A Setiawan, I. Made Agus
%Y Ben Abacha, Asma
%Y Bethard, Steven
%Y Bitterman, Danielle
%Y Naumann, Tristan
%Y Roberts, Kirk
%S Proceedings of the 8th Workshop on Clinical Natural Language Processing (Clinical NLP) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F aryoyudanta-etal-2026-night
%X We presented a small decoder-only language model for clinical observation extraction and normalization from nurse dictation developed using Reinforcement Learning with Verifiable Rewards (RLVR). We fine-tune Qwen3-1.7B model using a two-stage pipeline: (1) supervised fine-tuning (SFT) with an augmented chain-of-thought (CoT) dataset generated by a teacher model to mitigate RL cold-start, followed by (2) GRPO-based RLVR with multi-component reward functions that verify output format, concept presence, value type, and value correctness using the shared-task ontology (193 concepts) as a verifier. On the development set, SFT+GRPO substantially outperforms GRPO-only (F1 0.803 vs. 0.620). After the test holdout was released, our final system achieved 0.700 precision, 0.785 recall, and 0.740 F1. Error analysis shows remaining challenges in concept over-detection and missed concepts, as well as boundary errors in categorical and multi-select value type extraction. Our results demonstrates that small language models can enable accurate, cost-effective, and privacy preserving automated clinical documentation for nurse dictation, supporting scalable deployment in low-resource healthcare settings to reduce nurses’ documentation burden.
%R 10.63317/328obtnwpmsw
%U https://aclanthology.org/2026.clinicalnlp-1.18/
%U https://doi.org/10.63317/328obtnwpmsw
%P 163-173
Markdown (Informal)
[Night Shift Nerds at MEDIQA-SYNUR 2026: Pushing Small Large Language Model Capability for Clinical Observation Extraction and Normalization from Nurse Dictation using RLVR](https://aclanthology.org/2026.clinicalnlp-1.18/) (Aryoyudanta et al., ClinicalNLP 2026)
ACL