@inproceedings{duran-silva-etal-2026-normalizing,
title = "Normalizing Section Names and Structure of Scientific Articles",
author = "Duran-Silva, Nicolau and
Moreno-Schneider, Julian and
Parra-Rojas, C{\'e}sar A. and
Rehm, Georg",
editor = "Rehm, Georg and
Dietze, Stefan and
Dessi, Danilo and
Maynard, Diana and
Schimmler, Sonja",
booktitle = "Proceedings of Natural Scientific Language Processing ({NSLP}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nslp-1.21/",
doi = "10.63317/5ftk2fxxf7jd",
pages = "218--224",
abstract = "The growing amount of scientific literature has increased the need for automatic methods that can retrieve, process, and exploit scholarly content. In this work, we explore section name normalization and hierarchy prediction for scientific articles using a two-level taxonomy. We compare independent, sequential classification models, and generative large language models on the SASC dataset. Results show that classification approaches, particularly sequential models that employ document-level context, consistently outperform generative methods. Incorporating section content is essential for fine-grained classification, while generative models remain limited in zero-shot settings. Our experiments highlight the importance of structure-aware modelling for large-scale scholarly document processing, and the importance of section normalization for the development of advanced research mapping and research assessment tools."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="duran-silva-etal-2026-normalizing">
<titleInfo>
<title>Normalizing Section Names and Structure of Scientific Articles</title>
</titleInfo>
<name type="personal">
<namePart type="given">Nicolau</namePart>
<namePart type="family">Duran-Silva</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Julian</namePart>
<namePart type="family">Moreno-Schneider</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">César</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Parra-Rojas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stefan</namePart>
<namePart type="family">Dietze</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Danilo</namePart>
<namePart type="family">Dessi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Diana</namePart>
<namePart type="family">Maynard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sonja</namePart>
<namePart type="family">Schimmler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The growing amount of scientific literature has increased the need for automatic methods that can retrieve, process, and exploit scholarly content. In this work, we explore section name normalization and hierarchy prediction for scientific articles using a two-level taxonomy. We compare independent, sequential classification models, and generative large language models on the SASC dataset. Results show that classification approaches, particularly sequential models that employ document-level context, consistently outperform generative methods. Incorporating section content is essential for fine-grained classification, while generative models remain limited in zero-shot settings. Our experiments highlight the importance of structure-aware modelling for large-scale scholarly document processing, and the importance of section normalization for the development of advanced research mapping and research assessment tools.</abstract>
<identifier type="citekey">duran-silva-etal-2026-normalizing</identifier>
<identifier type="doi">10.63317/5ftk2fxxf7jd</identifier>
<location>
<url>https://aclanthology.org/2026.nslp-1.21/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>218</start>
<end>224</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Normalizing Section Names and Structure of Scientific Articles
%A Duran-Silva, Nicolau
%A Moreno-Schneider, Julian
%A Parra-Rojas, César A.
%A Rehm, Georg
%Y Rehm, Georg
%Y Dietze, Stefan
%Y Dessi, Danilo
%Y Maynard, Diana
%Y Schimmler, Sonja
%S Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F duran-silva-etal-2026-normalizing
%X The growing amount of scientific literature has increased the need for automatic methods that can retrieve, process, and exploit scholarly content. In this work, we explore section name normalization and hierarchy prediction for scientific articles using a two-level taxonomy. We compare independent, sequential classification models, and generative large language models on the SASC dataset. Results show that classification approaches, particularly sequential models that employ document-level context, consistently outperform generative methods. Incorporating section content is essential for fine-grained classification, while generative models remain limited in zero-shot settings. Our experiments highlight the importance of structure-aware modelling for large-scale scholarly document processing, and the importance of section normalization for the development of advanced research mapping and research assessment tools.
%R 10.63317/5ftk2fxxf7jd
%U https://aclanthology.org/2026.nslp-1.21/
%U https://doi.org/10.63317/5ftk2fxxf7jd
%P 218-224
Markdown (Informal)
[Normalizing Section Names and Structure of Scientific Articles](https://aclanthology.org/2026.nslp-1.21/) (Duran-Silva et al., NSLP 2026)
ACL
- Nicolau Duran-Silva, Julian Moreno-Schneider, César A. Parra-Rojas, and Georg Rehm. 2026. Normalizing Section Names and Structure of Scientific Articles. In Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026, pages 218–224, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).