@inproceedings{lee-etal-2026-automatic-evaluation,
title = "Automatic Evaluation of Multiple-Choice Items for Reading Comprehension: Effects of Question and Distractor Categories",
author = "Lee, John S. Y. and
Poon, Yin and
Wang, Shunjie and
Chu, Kai Wah",
editor = "Montejo-Raez, Arturo and
Grisot, Cristina and
Blochowiak, Joanna and
Ljube{\v{s}}i{\'c}, Nikola and
Battaner, Elena and
Rigau, German",
booktitle = "Proceedings of Shaping Multilingual, Multimodal {AI} for the Social Sciences and Humanities ({LLM}s4{SSH}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma de Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.llms4ssh-1.18/",
doi = "10.63317/2htm3v3vdbuv",
pages = "170--174",
abstract = "Automatic generation of multiple-choice (MC) items for reading comprehension can support language learning by providing large amounts of practice materials. To enable rapid development of MC generation models, automatic assessment is essential since it is time-consuming to manually evaluate question and distractor quality. Although Text Informativity (TI) has been adopted as an automatic evaluation metric, the ability of Large Language Models (LLMs) to estimate the TI scores of different categories of questions and distractors has not yet been thoroughly analyzed. This paper investigates LLM performance in calculating TI scores for the range of questions and distractors defined in the PIRLS (Progress in International Reading Literacy Study) and STARC (Structured Annotations for Reading Comprehension) frameworks. We show that automatically estimated TI scores may result in systematic preferences for some question and distractor categories, and recommend that TI scores be used for within-category comparisons only."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="lee-etal-2026-automatic-evaluation">
<titleInfo>
<title>Automatic Evaluation of Multiple-Choice Items for Reading Comprehension: Effects of Question and Distractor Categories</title>
</titleInfo>
<name type="personal">
<namePart type="given">John</namePart>
<namePart type="given">S</namePart>
<namePart type="given">Y</namePart>
<namePart type="family">Lee</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yin</namePart>
<namePart type="family">Poon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shunjie</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kai</namePart>
<namePart type="given">Wah</namePart>
<namePart type="family">Chu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Arturo</namePart>
<namePart type="family">Montejo-Raez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Cristina</namePart>
<namePart type="family">Grisot</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joanna</namePart>
<namePart type="family">Blochowiak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nikola</namePart>
<namePart type="family">Ljubešić</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Battaner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="family">Rigau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Automatic generation of multiple-choice (MC) items for reading comprehension can support language learning by providing large amounts of practice materials. To enable rapid development of MC generation models, automatic assessment is essential since it is time-consuming to manually evaluate question and distractor quality. Although Text Informativity (TI) has been adopted as an automatic evaluation metric, the ability of Large Language Models (LLMs) to estimate the TI scores of different categories of questions and distractors has not yet been thoroughly analyzed. This paper investigates LLM performance in calculating TI scores for the range of questions and distractors defined in the PIRLS (Progress in International Reading Literacy Study) and STARC (Structured Annotations for Reading Comprehension) frameworks. We show that automatically estimated TI scores may result in systematic preferences for some question and distractor categories, and recommend that TI scores be used for within-category comparisons only.</abstract>
<identifier type="citekey">lee-etal-2026-automatic-evaluation</identifier>
<identifier type="doi">10.63317/2htm3v3vdbuv</identifier>
<location>
<url>https://aclanthology.org/2026.llms4ssh-1.18/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>170</start>
<end>174</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Evaluation of Multiple-Choice Items for Reading Comprehension: Effects of Question and Distractor Categories
%A Lee, John S. Y.
%A Poon, Yin
%A Wang, Shunjie
%A Chu, Kai Wah
%Y Montejo-Raez, Arturo
%Y Grisot, Cristina
%Y Blochowiak, Joanna
%Y Ljubešić, Nikola
%Y Battaner, Elena
%Y Rigau, German
%S Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca (Spain)
%F lee-etal-2026-automatic-evaluation
%X Automatic generation of multiple-choice (MC) items for reading comprehension can support language learning by providing large amounts of practice materials. To enable rapid development of MC generation models, automatic assessment is essential since it is time-consuming to manually evaluate question and distractor quality. Although Text Informativity (TI) has been adopted as an automatic evaluation metric, the ability of Large Language Models (LLMs) to estimate the TI scores of different categories of questions and distractors has not yet been thoroughly analyzed. This paper investigates LLM performance in calculating TI scores for the range of questions and distractors defined in the PIRLS (Progress in International Reading Literacy Study) and STARC (Structured Annotations for Reading Comprehension) frameworks. We show that automatically estimated TI scores may result in systematic preferences for some question and distractor categories, and recommend that TI scores be used for within-category comparisons only.
%R 10.63317/2htm3v3vdbuv
%U https://aclanthology.org/2026.llms4ssh-1.18/
%U https://doi.org/10.63317/2htm3v3vdbuv
%P 170-174
Markdown (Informal)
[Automatic Evaluation of Multiple-Choice Items for Reading Comprehension: Effects of Question and Distractor Categories](https://aclanthology.org/2026.llms4ssh-1.18/) (Lee et al., LLMs4SSH 2026)
ACL