@inproceedings{mota-etal-2026-textbook,
title = "Textbook-Enriched Training for Language Models: Boosting Answer Quality in Specialized Contexts",
author = "Mota, Lucas B. Bulc{\~a}o and
Dantas, Larrissa and
Claro, Daniela Barreiro and
Paes, Aline and
Freitas, Claudia and
Souza, Marlo and
Caseli, Helena and
Real, Livy",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.23/",
doi = "10.5753/stil.2026.26566",
pages = "270--284",
abstract = "The use of textbooks as primary sources of information has increasingly given way to tools based on Large Language Models (LLMs), raising concerns about the reliability of generated answers. This study investigates how different adaptation strategies shape the behavior of small language models in educational Question Answering (QA) tasks in Portuguese. To support this analysis, we built a question-answer dataset derived from an NLP textbook and compared base models and the Retrieval-Augmented Generation (RAG) pipeline with models adapted through supervised fine-tuning and Continued Pretraining. The evaluation relies on questions from the LARI dataset, which has been validated by human specialists, and combines automatic and qualitative assessment procedures. The findings indicate that small models tuned with structured instructional knowledge achieve stronger semantic alignment and produce more pertinent answers in Portuguese educational QA scenarios."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mota-etal-2026-textbook">
<titleInfo>
<title>Textbook-Enriched Training for Language Models: Boosting Answer Quality in Specialized Contexts</title>
</titleInfo>
<name type="personal">
<namePart type="given">Lucas</namePart>
<namePart type="given">B</namePart>
<namePart type="given">Bulcão</namePart>
<namePart type="family">Mota</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Larrissa</namePart>
<namePart type="family">Dantas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Daniela</namePart>
<namePart type="given">Barreiro</namePart>
<namePart type="family">Claro</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Claudia</namePart>
<namePart type="family">Freitas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marlo</namePart>
<namePart type="family">Souza</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Caseli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Livy</namePart>
<namePart type="family">Real</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The use of textbooks as primary sources of information has increasingly given way to tools based on Large Language Models (LLMs), raising concerns about the reliability of generated answers. This study investigates how different adaptation strategies shape the behavior of small language models in educational Question Answering (QA) tasks in Portuguese. To support this analysis, we built a question-answer dataset derived from an NLP textbook and compared base models and the Retrieval-Augmented Generation (RAG) pipeline with models adapted through supervised fine-tuning and Continued Pretraining. The evaluation relies on questions from the LARI dataset, which has been validated by human specialists, and combines automatic and qualitative assessment procedures. The findings indicate that small models tuned with structured instructional knowledge achieve stronger semantic alignment and produce more pertinent answers in Portuguese educational QA scenarios.</abstract>
<identifier type="citekey">mota-etal-2026-textbook</identifier>
<identifier type="doi">10.5753/stil.2026.26566</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.23/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>270</start>
<end>284</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Textbook-Enriched Training for Language Models: Boosting Answer Quality in Specialized Contexts
%A Mota, Lucas B. Bulcão
%A Dantas, Larrissa
%A Claro, Daniela Barreiro
%A Paes, Aline
%A Freitas, Claudia
%A Souza, Marlo
%A Caseli, Helena
%A Real, Livy
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F mota-etal-2026-textbook
%X The use of textbooks as primary sources of information has increasingly given way to tools based on Large Language Models (LLMs), raising concerns about the reliability of generated answers. This study investigates how different adaptation strategies shape the behavior of small language models in educational Question Answering (QA) tasks in Portuguese. To support this analysis, we built a question-answer dataset derived from an NLP textbook and compared base models and the Retrieval-Augmented Generation (RAG) pipeline with models adapted through supervised fine-tuning and Continued Pretraining. The evaluation relies on questions from the LARI dataset, which has been validated by human specialists, and combines automatic and qualitative assessment procedures. The findings indicate that small models tuned with structured instructional knowledge achieve stronger semantic alignment and produce more pertinent answers in Portuguese educational QA scenarios.
%R 10.5753/stil.2026.26566
%U https://aclanthology.org/2026.stil-1.23/
%U https://doi.org/10.5753/stil.2026.26566
%P 270-284
Markdown (Informal)
[Textbook-Enriched Training for Language Models: Boosting Answer Quality in Specialized Contexts](https://aclanthology.org/2026.stil-1.23/) (Mota et al., STIL 2026)
ACL
- Lucas B. Bulcão Mota, Larrissa Dantas, Daniela Barreiro Claro, Aline Paes, Claudia Freitas, Marlo Souza, Helena Caseli, and Livy Real. 2026. Textbook-Enriched Training for Language Models: Boosting Answer Quality in Specialized Contexts. In Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology, pages 270–284, Cuiabá, Mato Grosso, Brazil. Association for Computational Linguistics.