@inproceedings{yildirim-etal-2026-automatic,
title = "Automatic Lemmatisation for {N}orwegian",
author = "Yildirim, Ahmet and
Hagen, Kristin and
Haug, Dag",
editor = "Hinrichs, Erhard and
Nivre, Joakim and
Osenova, Petya and
Pustejovsky, James and
Zinn, Claus",
booktitle = "Proceedings of the Workshop on Structured Linguistic Data and Evaluation ({SL}i{DE})",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.slide-1.8/",
doi = "10.63317/2cpfp5inka2c",
pages = "93--103",
abstract = "We report on a new lemmatisation system for Norwegian, which is a particularly challenging language with two written standards, Bokm{\r{a}}l and Nynorsk, that both have a lot of optionality. Our system covers both varieties and consists of a neural model that classifies words into rewrite rule classes that produce their lemma, as well as a large-scale computational lexicon of Norwegian that gives all possible inflections of a large part of the Norwegian vocabulary. We test different ways of combining these components. When evaluated with pure string-matching against the lemmas in the gold data, all systems perform approximately at the same level (99.1-99.2{\%} on Bokm{\r{a}}l and 98.5-98.6{\%} on Nynorsk), but detailed error analysis shows that the computational lexicon reduces the number of true errors by more than half (reaching 99.6{\%} accuracy on Bokm{\r{a}}l and 99.3{\%} on Nynorsk), as opposed to ``surface errors'' like using a different, but equally acceptable spelling variant of the correct lemma."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="yildirim-etal-2026-automatic">
<titleInfo>
<title>Automatic Lemmatisation for Norwegian</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ahmet</namePart>
<namePart type="family">Yildirim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kristin</namePart>
<namePart type="family">Hagen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dag</namePart>
<namePart type="family">Haug</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Workshop on Structured Linguistic Data and Evaluation (SLiDE)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Erhard</namePart>
<namePart type="family">Hinrichs</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joakim</namePart>
<namePart type="family">Nivre</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">James</namePart>
<namePart type="family">Pustejovsky</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Claus</namePart>
<namePart type="family">Zinn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We report on a new lemmatisation system for Norwegian, which is a particularly challenging language with two written standards, Bokmål and Nynorsk, that both have a lot of optionality. Our system covers both varieties and consists of a neural model that classifies words into rewrite rule classes that produce their lemma, as well as a large-scale computational lexicon of Norwegian that gives all possible inflections of a large part of the Norwegian vocabulary. We test different ways of combining these components. When evaluated with pure string-matching against the lemmas in the gold data, all systems perform approximately at the same level (99.1-99.2% on Bokmål and 98.5-98.6% on Nynorsk), but detailed error analysis shows that the computational lexicon reduces the number of true errors by more than half (reaching 99.6% accuracy on Bokmål and 99.3% on Nynorsk), as opposed to “surface errors” like using a different, but equally acceptable spelling variant of the correct lemma.</abstract>
<identifier type="citekey">yildirim-etal-2026-automatic</identifier>
<identifier type="doi">10.63317/2cpfp5inka2c</identifier>
<location>
<url>https://aclanthology.org/2026.slide-1.8/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>93</start>
<end>103</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Lemmatisation for Norwegian
%A Yildirim, Ahmet
%A Hagen, Kristin
%A Haug, Dag
%Y Hinrichs, Erhard
%Y Nivre, Joakim
%Y Osenova, Petya
%Y Pustejovsky, James
%Y Zinn, Claus
%S Proceedings of the Workshop on Structured Linguistic Data and Evaluation (SLiDE)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F yildirim-etal-2026-automatic
%X We report on a new lemmatisation system for Norwegian, which is a particularly challenging language with two written standards, Bokmål and Nynorsk, that both have a lot of optionality. Our system covers both varieties and consists of a neural model that classifies words into rewrite rule classes that produce their lemma, as well as a large-scale computational lexicon of Norwegian that gives all possible inflections of a large part of the Norwegian vocabulary. We test different ways of combining these components. When evaluated with pure string-matching against the lemmas in the gold data, all systems perform approximately at the same level (99.1-99.2% on Bokmål and 98.5-98.6% on Nynorsk), but detailed error analysis shows that the computational lexicon reduces the number of true errors by more than half (reaching 99.6% accuracy on Bokmål and 99.3% on Nynorsk), as opposed to “surface errors” like using a different, but equally acceptable spelling variant of the correct lemma.
%R 10.63317/2cpfp5inka2c
%U https://aclanthology.org/2026.slide-1.8/
%U https://doi.org/10.63317/2cpfp5inka2c
%P 93-103
Markdown (Informal)
[Automatic Lemmatisation for Norwegian](https://aclanthology.org/2026.slide-1.8/) (Yildirim et al., SLiDE 2026)
ACL
- Ahmet Yildirim, Kristin Hagen, and Dag Haug. 2026. Automatic Lemmatisation for Norwegian. In Proceedings of the Workshop on Structured Linguistic Data and Evaluation (SLiDE), pages 93–103, Palma de Mallorca, Spain. ELRA Language Resources Association (ELRA).