@inproceedings{hashimoto-shi-2026-politeness,
title = "Where Is Politeness in {J}apanese {BERT}? A Layerwise Probing and {CLS} Activation Patching Study",
author = "Hashimoto, Shusuke and
Shi, Wenchen",
editor = "Stranisci, Marco Antonio and
Falk, Neele and
Labat, Sofie and
Lo, Soda Marem and
Velutharambath, Aswathy and
Weber, Sabine and
Damiano, Rossana and
Frenda, Simona and
Hoste, Veronique and
Kleinberg, Bennett and
Klinger, Roman and
Patti, Viviana and
Plaza-del-Arco, Flor Miriam and
Sap, Maarten and
Yimam, Seid Muhie",
booktitle = "Proceedings of the 1st Workshop on Social Context ({S}o{C}on) and the 2nd Workshop on Integrating {NLP} and Psychology to Study Social Interactions ({NLPSI}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "European Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.socon-1.7/",
doi = "10.63317/4vex4yooz83k",
pages = "67--75",
abstract = "Politeness is a central pragmatic dimension of language use, and Japanese honorifics offer a well-defined testbed for studying whether pretrained encoders represent socially meaningful distinctions. Prior BERT-based work has applied supervised models to Japanese honorific data, but we are not aware of analyses that localize honorific-level information across layers or test causal influence via activation patching in Japanese BERT-style encoders. We study these questions in LineDistilBERT using the KeiCO corpus, which labels sentences with four honorific levels. To isolate pretrained representations while still defining a task predictor, we freeze all encoder parameters and train only a lightweight [CLS] classification head as a minimal readout. We then run layerwise linear probing, training multinomial L2-regularized logistic-regression probes on [CLS] vectors from each layer to quantify linear decodability across depth and to select a best layer on development data. Finally, we test causal leverage with [CLS] activation patching, transplanting donor activations into receiver sentences at selected layers and measuring prediction transitions, logit shifts, and flip rates under standard controls. Overall, honorific level is broadly decodable across layers, and [CLS] interventions can systematically steer the frozen-encoder classifier with strong depth dependence, providing complementary evidence from probing and causal intervention for Japanese politeness in practice."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hashimoto-shi-2026-politeness">
<titleInfo>
<title>Where Is Politeness in Japanese BERT? A Layerwise Probing and CLS Activation Patching Study</title>
</titleInfo>
<name type="personal">
<namePart type="given">Shusuke</namePart>
<namePart type="family">Hashimoto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wenchen</namePart>
<namePart type="family">Shi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 1st Workshop on Social Context (SoCon) and the 2nd Workshop on Integrating NLP and Psychology to Study Social Interactions (NLPSI) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="given">Antonio</namePart>
<namePart type="family">Stranisci</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Neele</namePart>
<namePart type="family">Falk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sofie</namePart>
<namePart type="family">Labat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Soda</namePart>
<namePart type="given">Marem</namePart>
<namePart type="family">Lo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aswathy</namePart>
<namePart type="family">Velutharambath</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sabine</namePart>
<namePart type="family">Weber</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rossana</namePart>
<namePart type="family">Damiano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simona</namePart>
<namePart type="family">Frenda</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Veronique</namePart>
<namePart type="family">Hoste</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bennett</namePart>
<namePart type="family">Kleinberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Roman</namePart>
<namePart type="family">Klinger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviana</namePart>
<namePart type="family">Patti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Flor</namePart>
<namePart type="given">Miriam</namePart>
<namePart type="family">Plaza-del-Arco</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maarten</namePart>
<namePart type="family">Sap</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Seid</namePart>
<namePart type="given">Muhie</namePart>
<namePart type="family">Yimam</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Politeness is a central pragmatic dimension of language use, and Japanese honorifics offer a well-defined testbed for studying whether pretrained encoders represent socially meaningful distinctions. Prior BERT-based work has applied supervised models to Japanese honorific data, but we are not aware of analyses that localize honorific-level information across layers or test causal influence via activation patching in Japanese BERT-style encoders. We study these questions in LineDistilBERT using the KeiCO corpus, which labels sentences with four honorific levels. To isolate pretrained representations while still defining a task predictor, we freeze all encoder parameters and train only a lightweight [CLS] classification head as a minimal readout. We then run layerwise linear probing, training multinomial L2-regularized logistic-regression probes on [CLS] vectors from each layer to quantify linear decodability across depth and to select a best layer on development data. Finally, we test causal leverage with [CLS] activation patching, transplanting donor activations into receiver sentences at selected layers and measuring prediction transitions, logit shifts, and flip rates under standard controls. Overall, honorific level is broadly decodable across layers, and [CLS] interventions can systematically steer the frozen-encoder classifier with strong depth dependence, providing complementary evidence from probing and causal intervention for Japanese politeness in practice.</abstract>
<identifier type="citekey">hashimoto-shi-2026-politeness</identifier>
<identifier type="doi">10.63317/4vex4yooz83k</identifier>
<location>
<url>https://aclanthology.org/2026.socon-1.7/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>67</start>
<end>75</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Where Is Politeness in Japanese BERT? A Layerwise Probing and CLS Activation Patching Study
%A Hashimoto, Shusuke
%A Shi, Wenchen
%Y Stranisci, Marco Antonio
%Y Falk, Neele
%Y Labat, Sofie
%Y Lo, Soda Marem
%Y Velutharambath, Aswathy
%Y Weber, Sabine
%Y Damiano, Rossana
%Y Frenda, Simona
%Y Hoste, Veronique
%Y Kleinberg, Bennett
%Y Klinger, Roman
%Y Patti, Viviana
%Y Plaza-del-Arco, Flor Miriam
%Y Sap, Maarten
%Y Yimam, Seid Muhie
%S Proceedings of the 1st Workshop on Social Context (SoCon) and the 2nd Workshop on Integrating NLP and Psychology to Study Social Interactions (NLPSI) @ LREC 2026
%D 2026
%8 May
%I European Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F hashimoto-shi-2026-politeness
%X Politeness is a central pragmatic dimension of language use, and Japanese honorifics offer a well-defined testbed for studying whether pretrained encoders represent socially meaningful distinctions. Prior BERT-based work has applied supervised models to Japanese honorific data, but we are not aware of analyses that localize honorific-level information across layers or test causal influence via activation patching in Japanese BERT-style encoders. We study these questions in LineDistilBERT using the KeiCO corpus, which labels sentences with four honorific levels. To isolate pretrained representations while still defining a task predictor, we freeze all encoder parameters and train only a lightweight [CLS] classification head as a minimal readout. We then run layerwise linear probing, training multinomial L2-regularized logistic-regression probes on [CLS] vectors from each layer to quantify linear decodability across depth and to select a best layer on development data. Finally, we test causal leverage with [CLS] activation patching, transplanting donor activations into receiver sentences at selected layers and measuring prediction transitions, logit shifts, and flip rates under standard controls. Overall, honorific level is broadly decodable across layers, and [CLS] interventions can systematically steer the frozen-encoder classifier with strong depth dependence, providing complementary evidence from probing and causal intervention for Japanese politeness in practice.
%R 10.63317/4vex4yooz83k
%U https://aclanthology.org/2026.socon-1.7/
%U https://doi.org/10.63317/4vex4yooz83k
%P 67-75
Markdown (Informal)
[Where Is Politeness in Japanese BERT? A Layerwise Probing and CLS Activation Patching Study](https://aclanthology.org/2026.socon-1.7/) (Hashimoto & Shi, SoCon-NLPSI 2026)
ACL