@inproceedings{kallas-etal-2026-using,
title = "Using {LLM}s to Extract Instances of Schematic Constructions from Unannotated {L}2 Learner Corpora",
author = "Kallas, Jelena and
Kiil, Ahto and
Sahkai, Heete and
Paulsen, Geda and
Saul, Kertu",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.824/",
doi = "10.63317/3ieeohd75wkj",
pages = "10517--10524",
abstract = "Our previous study found that generative LLMs can be successfully used to identify instances of schematic constructions (as defined in Construction Grammar) in unannotated L1 corpus data. This study tests the applicability of LLMs to also identify instances of constructions in unannotated L2 data. L2 learner corpora are notoriously difficult to annotate and query since they contain errors. Using LLMs can thus simplify the retrieval of construction data from L2 corpora. The identification of instances of constructions in L2 learner data has many possible uses in pedagogical applications of Construction Grammar and constructicography, like the identification of error-prone (properties of) constructions and the distribution of constructional instances across CEFR levels. Using the Estonian Nominal Quantifier Construction as the example construction and an Estonian CEFR-graded learner corpus as the source of L2 data, we tested several prompts and several models (OpenAI{'}s o3-mini, o3, gpt-5-mini and gpt-5, Google DeepMind{'}s Gemini Flash 2.5, Anthropic{'}s Claude Sonnet 4.5 and Opus 4.1). We found that the best model, gpt-5, achieved F1-scores from 0.90 to 0.96, depending on the level of detail of the prompt."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kallas-etal-2026-using">
<titleInfo>
<title>Using LLMs to Extract Instances of Schematic Constructions from Unannotated L2 Learner Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jelena</namePart>
<namePart type="family">Kallas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahto</namePart>
<namePart type="family">Kiil</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Heete</namePart>
<namePart type="family">Sahkai</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Geda</namePart>
<namePart type="family">Paulsen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kertu</namePart>
<namePart type="family">Saul</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Our previous study found that generative LLMs can be successfully used to identify instances of schematic constructions (as defined in Construction Grammar) in unannotated L1 corpus data. This study tests the applicability of LLMs to also identify instances of constructions in unannotated L2 data. L2 learner corpora are notoriously difficult to annotate and query since they contain errors. Using LLMs can thus simplify the retrieval of construction data from L2 corpora. The identification of instances of constructions in L2 learner data has many possible uses in pedagogical applications of Construction Grammar and constructicography, like the identification of error-prone (properties of) constructions and the distribution of constructional instances across CEFR levels. Using the Estonian Nominal Quantifier Construction as the example construction and an Estonian CEFR-graded learner corpus as the source of L2 data, we tested several prompts and several models (OpenAI’s o3-mini, o3, gpt-5-mini and gpt-5, Google DeepMind’s Gemini Flash 2.5, Anthropic’s Claude Sonnet 4.5 and Opus 4.1). We found that the best model, gpt-5, achieved F1-scores from 0.90 to 0.96, depending on the level of detail of the prompt.</abstract>
<identifier type="citekey">kallas-etal-2026-using</identifier>
<identifier type="doi">10.63317/3ieeohd75wkj</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.824/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>10517</start>
<end>10524</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Using LLMs to Extract Instances of Schematic Constructions from Unannotated L2 Learner Corpora
%A Kallas, Jelena
%A Kiil, Ahto
%A Sahkai, Heete
%A Paulsen, Geda
%A Saul, Kertu
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F kallas-etal-2026-using
%X Our previous study found that generative LLMs can be successfully used to identify instances of schematic constructions (as defined in Construction Grammar) in unannotated L1 corpus data. This study tests the applicability of LLMs to also identify instances of constructions in unannotated L2 data. L2 learner corpora are notoriously difficult to annotate and query since they contain errors. Using LLMs can thus simplify the retrieval of construction data from L2 corpora. The identification of instances of constructions in L2 learner data has many possible uses in pedagogical applications of Construction Grammar and constructicography, like the identification of error-prone (properties of) constructions and the distribution of constructional instances across CEFR levels. Using the Estonian Nominal Quantifier Construction as the example construction and an Estonian CEFR-graded learner corpus as the source of L2 data, we tested several prompts and several models (OpenAI’s o3-mini, o3, gpt-5-mini and gpt-5, Google DeepMind’s Gemini Flash 2.5, Anthropic’s Claude Sonnet 4.5 and Opus 4.1). We found that the best model, gpt-5, achieved F1-scores from 0.90 to 0.96, depending on the level of detail of the prompt.
%R 10.63317/3ieeohd75wkj
%U https://aclanthology.org/2026.lrec-1.824/
%U https://doi.org/10.63317/3ieeohd75wkj
%P 10517-10524
Markdown (Informal)
[Using LLMs to Extract Instances of Schematic Constructions from Unannotated L2 Learner Corpora](https://aclanthology.org/2026.lrec-1.824/) (Kallas et al., LREC 2026)
ACL