@article{wang-etal-2026-explicit,
title = "From Explicit to Implicit: A Theoretical Framework and Transfer Method for Preference Internalization in Language Models",
author = "Wang, Binrui and
Du, Yongping and
Pei, Yu and
Wang, Zikai",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.48/",
doi = "10.1162/tacl.a.697",
pages = "1074--1095",
abstract = "Transforming explicit preference signals into implicit and parameterized behaviors is pivotal for enabling prompt-free, human-aligned generation and improving the usability, efficiency and robustness of large language models. However, existing methods align model preference well but still rely on explicit user instructions to convey specific preferences, leading to cumbersome user experiences and undermining natural, frictionless interaction with the model. To fill the gap between explicit and implicit preference representation, this paper introduces a theoretical framework that establishes both necessary and sufficient conditions for effective preference recognition. Based on this framework, we propose a novel Two-Stage Progressive Preference Transfer (TSPPT) method, which decomposes preference internalization into two manageable stages: preference representation learning and preference internalization transfer. The proposed method fills the gap between explicit and implicit preferences while maintaining the model{'}s general capabilities. The experiments across multiple models (Qwen2.5, Qwen3, Llama-3.2, DeepSeek-R1-Distill) and datasets (UltraFeedback, HelpSteer) demonstrate superior performance. The proposed method achieves 79.2{\%} win rate on UltraFeedback (vs. 59.2{--}67.6{\%} for baselines), substantial improvements on MT-Bench (7.86 vs. 7.34 for best baseline), and significant reductions in implicit social bias (0.165 vs. 0.185{--}0.325 for baselines). Notably, the method maintains comparable performance between implicit and explicit settings, confirming successful preference internalization.1"
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="wang-etal-2026-explicit">
<titleInfo>
<title>From Explicit to Implicit: A Theoretical Framework and Transfer Method for Preference Internalization in Language Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Binrui</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yongping</namePart>
<namePart type="family">Du</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yu</namePart>
<namePart type="family">Pei</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zikai</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Transforming explicit preference signals into implicit and parameterized behaviors is pivotal for enabling prompt-free, human-aligned generation and improving the usability, efficiency and robustness of large language models. However, existing methods align model preference well but still rely on explicit user instructions to convey specific preferences, leading to cumbersome user experiences and undermining natural, frictionless interaction with the model. To fill the gap between explicit and implicit preference representation, this paper introduces a theoretical framework that establishes both necessary and sufficient conditions for effective preference recognition. Based on this framework, we propose a novel Two-Stage Progressive Preference Transfer (TSPPT) method, which decomposes preference internalization into two manageable stages: preference representation learning and preference internalization transfer. The proposed method fills the gap between explicit and implicit preferences while maintaining the model’s general capabilities. The experiments across multiple models (Qwen2.5, Qwen3, Llama-3.2, DeepSeek-R1-Distill) and datasets (UltraFeedback, HelpSteer) demonstrate superior performance. The proposed method achieves 79.2% win rate on UltraFeedback (vs. 59.2–67.6% for baselines), substantial improvements on MT-Bench (7.86 vs. 7.34 for best baseline), and significant reductions in implicit social bias (0.165 vs. 0.185–0.325 for baselines). Notably, the method maintains comparable performance between implicit and explicit settings, confirming successful preference internalization.1</abstract>
<identifier type="citekey">wang-etal-2026-explicit</identifier>
<identifier type="doi">10.1162/tacl.a.697</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.48/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>1074</start>
<end>1095</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T From Explicit to Implicit: A Theoretical Framework and Transfer Method for Preference Internalization in Language Models
%A Wang, Binrui
%A Du, Yongping
%A Pei, Yu
%A Wang, Zikai
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F wang-etal-2026-explicit
%X Transforming explicit preference signals into implicit and parameterized behaviors is pivotal for enabling prompt-free, human-aligned generation and improving the usability, efficiency and robustness of large language models. However, existing methods align model preference well but still rely on explicit user instructions to convey specific preferences, leading to cumbersome user experiences and undermining natural, frictionless interaction with the model. To fill the gap between explicit and implicit preference representation, this paper introduces a theoretical framework that establishes both necessary and sufficient conditions for effective preference recognition. Based on this framework, we propose a novel Two-Stage Progressive Preference Transfer (TSPPT) method, which decomposes preference internalization into two manageable stages: preference representation learning and preference internalization transfer. The proposed method fills the gap between explicit and implicit preferences while maintaining the model’s general capabilities. The experiments across multiple models (Qwen2.5, Qwen3, Llama-3.2, DeepSeek-R1-Distill) and datasets (UltraFeedback, HelpSteer) demonstrate superior performance. The proposed method achieves 79.2% win rate on UltraFeedback (vs. 59.2–67.6% for baselines), substantial improvements on MT-Bench (7.86 vs. 7.34 for best baseline), and significant reductions in implicit social bias (0.165 vs. 0.185–0.325 for baselines). Notably, the method maintains comparable performance between implicit and explicit settings, confirming successful preference internalization.1
%R 10.1162/tacl.a.697
%U https://aclanthology.org/2026.tacl-1.48/
%U https://doi.org/10.1162/tacl.a.697
%P 1074-1095
Markdown (Informal)
[From Explicit to Implicit: A Theoretical Framework and Transfer Method for Preference Internalization in Language Models](https://aclanthology.org/2026.tacl-1.48/) (Wang et al., TACL 2026)
ACL