@inproceedings{suresh-etal-2026-flag,
title = "To Flag or Not to Flag? Detecting Concerning Content in Situational Judgment",
author = "Suresh, Susha and
Walsh, Cole and
Ivan, Rodica and
Robb, Colleen",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.50/",
pages = "450--457",
ISBN = "979-8-9983004-0-0",
abstract = "This study evaluates automated approaches for detecting concerning content in open-response SJTs used in higher education admissions. Comparing fine-tuned BERT models with zero-shot and fine-tuned LLMs, we found that fine-tuned BERT achieved the strongest performance despite not receiving the scenario context available to the LLMs"
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="suresh-etal-2026-flag">
<titleInfo>
<title>To Flag or Not to Flag? Detecting Concerning Content in Situational Judgment</title>
</titleInfo>
<name type="personal">
<namePart type="given">Susha</namePart>
<namePart type="family">Suresh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Cole</namePart>
<namePart type="family">Walsh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rodica</namePart>
<namePart type="family">Ivan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Colleen</namePart>
<namePart type="family">Robb</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study evaluates automated approaches for detecting concerning content in open-response SJTs used in higher education admissions. Comparing fine-tuned BERT models with zero-shot and fine-tuned LLMs, we found that fine-tuned BERT achieved the strongest performance despite not receiving the scenario context available to the LLMs</abstract>
<identifier type="citekey">suresh-etal-2026-flag</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.50/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>450</start>
<end>457</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T To Flag or Not to Flag? Detecting Concerning Content in Situational Judgment
%A Suresh, Susha
%A Walsh, Cole
%A Ivan, Rodica
%A Robb, Colleen
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F suresh-etal-2026-flag
%X This study evaluates automated approaches for detecting concerning content in open-response SJTs used in higher education admissions. Comparing fine-tuned BERT models with zero-shot and fine-tuned LLMs, we found that fine-tuned BERT achieved the strongest performance despite not receiving the scenario context available to the LLMs
%U https://aclanthology.org/2026.aimecon-main.50/
%P 450-457
Markdown (Informal)
[To Flag or Not to Flag? Detecting Concerning Content in Situational Judgment](https://aclanthology.org/2026.aimecon-main.50/) (Suresh et al., AIME-Con 2026)
ACL
- Susha Suresh, Cole Walsh, Rodica Ivan, and Colleen Robb. 2026. To Flag or Not to Flag? Detecting Concerning Content in Situational Judgment. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers, pages 450–457, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).