@inproceedings{osumi-etal-2026-examining,
title = "Examining {LLM}-Surprisal as an Indicator of Naturalness for {J}apanese Automated Essay Scoring",
author = "Osumi, Akari and
Hu, Jingying and
Cong, Yan and
Fukada, Atsushi",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.24/",
pages = "221--229",
ISBN = "979-8-9983004-0-0",
abstract = "This study investigates Large Language Model (LLM) surprisal as an indicator of global linguistic naturalness in L2 Japanese automatic essay scoring. Results demonstrate that surprisal effectively distinguishes learner proficiency levels. Combining surprisal with feature-based indices achieves the highest classification accuracy, validating surprisal as a valuable metric for automated writing evaluation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="osumi-etal-2026-examining">
<titleInfo>
<title>Examining LLM-Surprisal as an Indicator of Naturalness for Japanese Automated Essay Scoring</title>
</titleInfo>
<name type="personal">
<namePart type="given">Akari</namePart>
<namePart type="family">Osumi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jingying</namePart>
<namePart type="family">Hu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yan</namePart>
<namePart type="family">Cong</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Atsushi</namePart>
<namePart type="family">Fukada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study investigates Large Language Model (LLM) surprisal as an indicator of global linguistic naturalness in L2 Japanese automatic essay scoring. Results demonstrate that surprisal effectively distinguishes learner proficiency levels. Combining surprisal with feature-based indices achieves the highest classification accuracy, validating surprisal as a valuable metric for automated writing evaluation.</abstract>
<identifier type="citekey">osumi-etal-2026-examining</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.24/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>221</start>
<end>229</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Examining LLM-Surprisal as an Indicator of Naturalness for Japanese Automated Essay Scoring
%A Osumi, Akari
%A Hu, Jingying
%A Cong, Yan
%A Fukada, Atsushi
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F osumi-etal-2026-examining
%X This study investigates Large Language Model (LLM) surprisal as an indicator of global linguistic naturalness in L2 Japanese automatic essay scoring. Results demonstrate that surprisal effectively distinguishes learner proficiency levels. Combining surprisal with feature-based indices achieves the highest classification accuracy, validating surprisal as a valuable metric for automated writing evaluation.
%U https://aclanthology.org/2026.aimecon-main.24/
%P 221-229
Markdown (Informal)
[Examining LLM-Surprisal as an Indicator of Naturalness for Japanese Automated Essay Scoring](https://aclanthology.org/2026.aimecon-main.24/) (Osumi et al., AIME-Con 2026)
ACL