@inproceedings{kaushik-anand-2026-finerviner,
title = "{F}i{NERVINER}: Fine-grained Named Entity Recognition for Vulnerable Languages of {I}ndia{'}s North Eastern Region",
author = "Kaushik, Prachuryya and
Anand, Ashish",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.607/",
doi = "10.63317/3rs5mcedzvss",
pages = "7655--7667",
abstract = "Named entity recognition (NER), particularly fine-grained NER (FgNER), extracts domain-specific entity information for Natural Language Processing (NLP) applications such as knowledge base construction and relation extraction. While manual annotation for creating relevant data is expensive, distant supervision often produces noisy data. Moreover, resources for coarse-grained and fine-grained NER in Indian languages, particularly in the vulnerable languages of India{'}s North Eastern Region, remain scarce. This work aims at creating such a resource for three vulnerable languages: {\ensuremath{<}}i{\ensuremath{>}}Bodo/Boro (brx){\ensuremath{<}}/i{\ensuremath{>}}, {\ensuremath{<}}i{\ensuremath{>}}Manipuri/Meitei (mni){\ensuremath{<}}/i{\ensuremath{>}}, and {\ensuremath{<}}i{\ensuremath{>}}Mizo/Lushai (lus){\ensuremath{<}}/i{\ensuremath{>}}, which are regarded as official languages in three Indian states and spoken by more than six million people across five countries in South and Southeast Asia. We use annotations projection from high-resource FgNER datasets using source-to-target parallel corpora and a projection tool built on a multilingual encoder. The dataset comprises over 198k sentences, 282k entities, and 2.8M tokens in each low-resource language. Our thorough analyses validate the dataset{'}s high quality. We further explore zero-shot and cross-lingual settings, examining the impact of script similarity and multilingualism in cross-lingual FgNER performance. The dataset, expert detector models, the agentic tool, and the interactive web application are available as open-source resources at: \url{https://hf.co/collections/prachuryyaIITG/finerviner}."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kaushik-anand-2026-finerviner">
<titleInfo>
<title>FiNERVINER: Fine-grained Named Entity Recognition for Vulnerable Languages of India’s North Eastern Region</title>
</titleInfo>
<name type="personal">
<namePart type="given">Prachuryya</namePart>
<namePart type="family">Kaushik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ashish</namePart>
<namePart type="family">Anand</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Named entity recognition (NER), particularly fine-grained NER (FgNER), extracts domain-specific entity information for Natural Language Processing (NLP) applications such as knowledge base construction and relation extraction. While manual annotation for creating relevant data is expensive, distant supervision often produces noisy data. Moreover, resources for coarse-grained and fine-grained NER in Indian languages, particularly in the vulnerable languages of India’s North Eastern Region, remain scarce. This work aims at creating such a resource for three vulnerable languages: \ensuremath<i\ensuremath>Bodo/Boro (brx)\ensuremath</i\ensuremath>, \ensuremath<i\ensuremath>Manipuri/Meitei (mni)\ensuremath</i\ensuremath>, and \ensuremath<i\ensuremath>Mizo/Lushai (lus)\ensuremath</i\ensuremath>, which are regarded as official languages in three Indian states and spoken by more than six million people across five countries in South and Southeast Asia. We use annotations projection from high-resource FgNER datasets using source-to-target parallel corpora and a projection tool built on a multilingual encoder. The dataset comprises over 198k sentences, 282k entities, and 2.8M tokens in each low-resource language. Our thorough analyses validate the dataset’s high quality. We further explore zero-shot and cross-lingual settings, examining the impact of script similarity and multilingualism in cross-lingual FgNER performance. The dataset, expert detector models, the agentic tool, and the interactive web application are available as open-source resources at: https://hf.co/collections/prachuryyaIITG/finerviner.</abstract>
<identifier type="citekey">kaushik-anand-2026-finerviner</identifier>
<identifier type="doi">10.63317/3rs5mcedzvss</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.607/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>7655</start>
<end>7667</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T FiNERVINER: Fine-grained Named Entity Recognition for Vulnerable Languages of India’s North Eastern Region
%A Kaushik, Prachuryya
%A Anand, Ashish
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F kaushik-anand-2026-finerviner
%X Named entity recognition (NER), particularly fine-grained NER (FgNER), extracts domain-specific entity information for Natural Language Processing (NLP) applications such as knowledge base construction and relation extraction. While manual annotation for creating relevant data is expensive, distant supervision often produces noisy data. Moreover, resources for coarse-grained and fine-grained NER in Indian languages, particularly in the vulnerable languages of India’s North Eastern Region, remain scarce. This work aims at creating such a resource for three vulnerable languages: \ensuremath<i\ensuremath>Bodo/Boro (brx)\ensuremath</i\ensuremath>, \ensuremath<i\ensuremath>Manipuri/Meitei (mni)\ensuremath</i\ensuremath>, and \ensuremath<i\ensuremath>Mizo/Lushai (lus)\ensuremath</i\ensuremath>, which are regarded as official languages in three Indian states and spoken by more than six million people across five countries in South and Southeast Asia. We use annotations projection from high-resource FgNER datasets using source-to-target parallel corpora and a projection tool built on a multilingual encoder. The dataset comprises over 198k sentences, 282k entities, and 2.8M tokens in each low-resource language. Our thorough analyses validate the dataset’s high quality. We further explore zero-shot and cross-lingual settings, examining the impact of script similarity and multilingualism in cross-lingual FgNER performance. The dataset, expert detector models, the agentic tool, and the interactive web application are available as open-source resources at: https://hf.co/collections/prachuryyaIITG/finerviner.
%R 10.63317/3rs5mcedzvss
%U https://aclanthology.org/2026.lrec-1.607/
%U https://doi.org/10.63317/3rs5mcedzvss
%P 7655-7667
Markdown (Informal)
[FiNERVINER: Fine-grained Named Entity Recognition for Vulnerable Languages of India’s North Eastern Region](https://aclanthology.org/2026.lrec-1.607/) (Kaushik & Anand, LREC 2026)
ACL