@inproceedings{freitas-berton-2026-epicorpus,
title = "{E}pi{C}orpus-{BR}: A Named Entity Recognition Corpus for Epidemiological Documents in {P}ortuguese",
author = "Freitas, Christian and
Berton, Lilian",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.13/",
doi = "10.5753/stil.2026.26614",
pages = "150--162",
abstract = "This paper presents EpiCorpus-BR, the first annotated corpus for Named Entity Recognition (NER) in Brazilian Portuguese epidemiological surveillance documents, covering the annual SIREVA-SUS reports (2013{--}2024) and complementary documents from the Brazilian Ministry of Health. The corpus comprises 8,314 textual units across 22 documents (6,358 narrative text sentences and 1,956 table rows), with 4,159 entity mentions in the silver set. The tagset consists of eight specialized categories (PATOGENO, SOROTIPO, FAIXA{\_}ETARIA, LOCAL, PERIODO{\_}TEMPORAL, METODO{\_}LAB, METRICA{\_}EPI, MANIFESTACAO{\_}CLINICA), annotated via few-shot prompting with GPT-4o-mini and manually reviewed in a stratified sample of 280 units (180 text sentences and 100 table rows) by two independent annotators ({\ensuremath{\kappa}} = 0.76). Strict IOB2 evaluation with seqeval yields a combined micro-F1 of 0.77. METRICA{\_}EPI is absent from narrative text but reaches F1 = 0.89 on table rows, which shows why the tabular sub-corpus must be included in the evaluation. A BERTimbau baseline fine-tuned on the silver corpus, with the gold held out, reaches micro-F1 = 0.73, showing that the corpus supports training a dedicated NER model. The corpus and code are publicly available."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="freitas-berton-2026-epicorpus">
<titleInfo>
<title>EpiCorpus-BR: A Named Entity Recognition Corpus for Epidemiological Documents in Portuguese</title>
</titleInfo>
<name type="personal">
<namePart type="given">Christian</namePart>
<namePart type="family">Freitas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lilian</namePart>
<namePart type="family">Berton</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper presents EpiCorpus-BR, the first annotated corpus for Named Entity Recognition (NER) in Brazilian Portuguese epidemiological surveillance documents, covering the annual SIREVA-SUS reports (2013–2024) and complementary documents from the Brazilian Ministry of Health. The corpus comprises 8,314 textual units across 22 documents (6,358 narrative text sentences and 1,956 table rows), with 4,159 entity mentions in the silver set. The tagset consists of eight specialized categories (PATOGENO, SOROTIPO, FAIXA_ETARIA, LOCAL, PERIODO_TEMPORAL, METODO_LAB, METRICA_EPI, MANIFESTACAO_CLINICA), annotated via few-shot prompting with GPT-4o-mini and manually reviewed in a stratified sample of 280 units (180 text sentences and 100 table rows) by two independent annotators (\ensuremathąppa = 0.76). Strict IOB2 evaluation with seqeval yields a combined micro-F1 of 0.77. METRICA_EPI is absent from narrative text but reaches F1 = 0.89 on table rows, which shows why the tabular sub-corpus must be included in the evaluation. A BERTimbau baseline fine-tuned on the silver corpus, with the gold held out, reaches micro-F1 = 0.73, showing that the corpus supports training a dedicated NER model. The corpus and code are publicly available.</abstract>
<identifier type="citekey">freitas-berton-2026-epicorpus</identifier>
<identifier type="doi">10.5753/stil.2026.26614</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.13/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>150</start>
<end>162</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T EpiCorpus-BR: A Named Entity Recognition Corpus for Epidemiological Documents in Portuguese
%A Freitas, Christian
%A Berton, Lilian
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F freitas-berton-2026-epicorpus
%X This paper presents EpiCorpus-BR, the first annotated corpus for Named Entity Recognition (NER) in Brazilian Portuguese epidemiological surveillance documents, covering the annual SIREVA-SUS reports (2013–2024) and complementary documents from the Brazilian Ministry of Health. The corpus comprises 8,314 textual units across 22 documents (6,358 narrative text sentences and 1,956 table rows), with 4,159 entity mentions in the silver set. The tagset consists of eight specialized categories (PATOGENO, SOROTIPO, FAIXA_ETARIA, LOCAL, PERIODO_TEMPORAL, METODO_LAB, METRICA_EPI, MANIFESTACAO_CLINICA), annotated via few-shot prompting with GPT-4o-mini and manually reviewed in a stratified sample of 280 units (180 text sentences and 100 table rows) by two independent annotators (\ensuremathąppa = 0.76). Strict IOB2 evaluation with seqeval yields a combined micro-F1 of 0.77. METRICA_EPI is absent from narrative text but reaches F1 = 0.89 on table rows, which shows why the tabular sub-corpus must be included in the evaluation. A BERTimbau baseline fine-tuned on the silver corpus, with the gold held out, reaches micro-F1 = 0.73, showing that the corpus supports training a dedicated NER model. The corpus and code are publicly available.
%R 10.5753/stil.2026.26614
%U https://aclanthology.org/2026.stil-1.13/
%U https://doi.org/10.5753/stil.2026.26614
%P 150-162
Markdown (Informal)
[EpiCorpus-BR: A Named Entity Recognition Corpus for Epidemiological Documents in Portuguese](https://aclanthology.org/2026.stil-1.13/) (Freitas & Berton, STIL 2026)
ACL