@inproceedings{nasim-etal-2026-evaluating,
title = "Evaluating Large Language Models for Medical Named Entity Recognition in {U}rdu: A Benchmark Study",
author = "Nasim, Bushra and
Latif, Kinza and
Zohair, Muhammad and
Asif, Muhammad Hassan and
Nasim, Zarmeen",
editor = "Sarveswaran, Kengatharaiyer and
Vaidya, Ashwini",
booktitle = "Proceedings of the Second workshop on Challenges in Processing {S}outh {A}sian Languages ({CH}i{PSAL}2026)",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.chipsal-1.5/",
doi = "10.63317/5iobohdgnfh4",
pages = "40--48",
abstract = "Medical named entity recognition (NER) is a crucial task in natural language processing (NLP) for extracting meaningful entities such as diseases, symptoms, medications, body parts, and treatments from clinical text. However, NER in low-resource languages like Urdu remains underexplored due to limited annotated datasets. In this study, we evaluated the performance of two state-of-the-art large language models (LLMs), ChatGPT-4o and LLAMA 3.2, on Urdu medical NER using a dataset of 2,057 health-related Urdu news headlines manually annotated across five entity categories. Both models were evaluated using precision, recall, and F1-score. It was found that both models exhibited low precision and moderate recall. ChatGPT-4o achieved the highest F1 for Disease (0.35) while LLAMA 3.2 reached slightly lower F1 scores for Disease (0.33). Both models performed poorly on treatment-related terms, with F1 scores of 0.036 (LLAMA 3.2) and 0.011 (ChatGPT-4o). Micro-average F1-scores were 0.187 for ChatGPT-4o and 0.183 for LLAMA 3.2, indicating comparable overall performance. These findings highlight the challenges of medical NER in low-resource languages and underscore the need for domain-specific fine-tuning, transfer learning, few-shot learning, and prompt engineering to improve performance."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="nasim-etal-2026-evaluating">
<titleInfo>
<title>Evaluating Large Language Models for Medical Named Entity Recognition in Urdu: A Benchmark Study</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bushra</namePart>
<namePart type="family">Nasim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kinza</namePart>
<namePart type="family">Latif</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Muhammad</namePart>
<namePart type="family">Zohair</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Muhammad</namePart>
<namePart type="given">Hassan</namePart>
<namePart type="family">Asif</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zarmeen</namePart>
<namePart type="family">Nasim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second workshop on Challenges in Processing South Asian Languages (CHiPSAL2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kengatharaiyer</namePart>
<namePart type="family">Sarveswaran</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ashwini</namePart>
<namePart type="family">Vaidya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Medical named entity recognition (NER) is a crucial task in natural language processing (NLP) for extracting meaningful entities such as diseases, symptoms, medications, body parts, and treatments from clinical text. However, NER in low-resource languages like Urdu remains underexplored due to limited annotated datasets. In this study, we evaluated the performance of two state-of-the-art large language models (LLMs), ChatGPT-4o and LLAMA 3.2, on Urdu medical NER using a dataset of 2,057 health-related Urdu news headlines manually annotated across five entity categories. Both models were evaluated using precision, recall, and F1-score. It was found that both models exhibited low precision and moderate recall. ChatGPT-4o achieved the highest F1 for Disease (0.35) while LLAMA 3.2 reached slightly lower F1 scores for Disease (0.33). Both models performed poorly on treatment-related terms, with F1 scores of 0.036 (LLAMA 3.2) and 0.011 (ChatGPT-4o). Micro-average F1-scores were 0.187 for ChatGPT-4o and 0.183 for LLAMA 3.2, indicating comparable overall performance. These findings highlight the challenges of medical NER in low-resource languages and underscore the need for domain-specific fine-tuning, transfer learning, few-shot learning, and prompt engineering to improve performance.</abstract>
<identifier type="citekey">nasim-etal-2026-evaluating</identifier>
<identifier type="doi">10.63317/5iobohdgnfh4</identifier>
<location>
<url>https://aclanthology.org/2026.chipsal-1.5/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>40</start>
<end>48</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating Large Language Models for Medical Named Entity Recognition in Urdu: A Benchmark Study
%A Nasim, Bushra
%A Latif, Kinza
%A Zohair, Muhammad
%A Asif, Muhammad Hassan
%A Nasim, Zarmeen
%Y Sarveswaran, Kengatharaiyer
%Y Vaidya, Ashwini
%S Proceedings of the Second workshop on Challenges in Processing South Asian Languages (CHiPSAL2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F nasim-etal-2026-evaluating
%X Medical named entity recognition (NER) is a crucial task in natural language processing (NLP) for extracting meaningful entities such as diseases, symptoms, medications, body parts, and treatments from clinical text. However, NER in low-resource languages like Urdu remains underexplored due to limited annotated datasets. In this study, we evaluated the performance of two state-of-the-art large language models (LLMs), ChatGPT-4o and LLAMA 3.2, on Urdu medical NER using a dataset of 2,057 health-related Urdu news headlines manually annotated across five entity categories. Both models were evaluated using precision, recall, and F1-score. It was found that both models exhibited low precision and moderate recall. ChatGPT-4o achieved the highest F1 for Disease (0.35) while LLAMA 3.2 reached slightly lower F1 scores for Disease (0.33). Both models performed poorly on treatment-related terms, with F1 scores of 0.036 (LLAMA 3.2) and 0.011 (ChatGPT-4o). Micro-average F1-scores were 0.187 for ChatGPT-4o and 0.183 for LLAMA 3.2, indicating comparable overall performance. These findings highlight the challenges of medical NER in low-resource languages and underscore the need for domain-specific fine-tuning, transfer learning, few-shot learning, and prompt engineering to improve performance.
%R 10.63317/5iobohdgnfh4
%U https://aclanthology.org/2026.chipsal-1.5/
%U https://doi.org/10.63317/5iobohdgnfh4
%P 40-48
Markdown (Informal)
[Evaluating Large Language Models for Medical Named Entity Recognition in Urdu: A Benchmark Study](https://aclanthology.org/2026.chipsal-1.5/) (Nasim et al., CHiPSAL 2026)
ACL