@inproceedings{loiseau-etal-2026-distilling,
title = "Distilling Human-Aligned Privacy Sensitivity Assessment from Large Language Models",
author = "Loiseau, Gabriel and
Sileo, Damien and
Riquet, Damien and
Meyer, Maxime and
Tommasi, Marc",
editor = {Siegert, Ingo and
Szawerna, Maria Irena and
Choukri, Khalid and
Dobnik, Simon and
Kamocki, Pawe{\l} and
Lindstr{\"o}m Tiedemann, Therese and
Lison, Pierre and
Mu{\~n}oz S{\'a}nchez, Ricardo and
Pil{\'a}n, Ildik{\'o} and
S{\"o}derg{\r{a}}rd, Lisa and
Talmoudi, Kossay and
Volodina, Elena and
Vu, Xuan-Son},
booktitle = "Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy ({LEGAL}2026 and {CALD}-pseudo 2026) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA",
url = "https://aclanthology.org/2026.legal-1.6/",
doi = "10.63317/3kv2f8gmz5rb",
pages = "53--61",
abstract = "Accurate privacy evaluation of textual data remains a critical challenge in privacy-preserving NLP. Recent work has shown that LLMs can serve as reliable privacy evaluators, achieving strong agreement with human judgments; however, their computational cost and impracticality for processing sensitive data at scale limit real-world deployment. We address this gap by distilling the privacy assessment capabilities of Mistral Large 3 (675B) into lightweight encoder models with as few as 150M parameters. Leveraging a large-scale dataset of privacy-annotated texts spanning 10 diverse domains, we train efficient classifiers that preserve strong agreement with human annotations while dramatically reducing computational requirements. We validate our approach on human-annotated test data and demonstrate its practical utility as an evaluation metric for de-identification systems."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="loiseau-etal-2026-distilling">
<titleInfo>
<title>Distilling Human-Aligned Privacy Sensitivity Assessment from Large Language Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gabriel</namePart>
<namePart type="family">Loiseau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Damien</namePart>
<namePart type="family">Sileo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Damien</namePart>
<namePart type="family">Riquet</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maxime</namePart>
<namePart type="family">Meyer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Tommasi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy (LEGAL2026 and CALD-pseudo 2026) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ingo</namePart>
<namePart type="family">Siegert</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="given">Irena</namePart>
<namePart type="family">Szawerna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalid</namePart>
<namePart type="family">Choukri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Dobnik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paweł</namePart>
<namePart type="family">Kamocki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Therese</namePart>
<namePart type="family">Lindström Tiedemann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Lison</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ricardo</namePart>
<namePart type="family">Muñoz Sánchez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ildikó</namePart>
<namePart type="family">Pilán</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Södergård</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kossay</namePart>
<namePart type="family">Talmoudi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Volodina</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xuan-Son</namePart>
<namePart type="family">Vu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Accurate privacy evaluation of textual data remains a critical challenge in privacy-preserving NLP. Recent work has shown that LLMs can serve as reliable privacy evaluators, achieving strong agreement with human judgments; however, their computational cost and impracticality for processing sensitive data at scale limit real-world deployment. We address this gap by distilling the privacy assessment capabilities of Mistral Large 3 (675B) into lightweight encoder models with as few as 150M parameters. Leveraging a large-scale dataset of privacy-annotated texts spanning 10 diverse domains, we train efficient classifiers that preserve strong agreement with human annotations while dramatically reducing computational requirements. We validate our approach on human-annotated test data and demonstrate its practical utility as an evaluation metric for de-identification systems.</abstract>
<identifier type="citekey">loiseau-etal-2026-distilling</identifier>
<identifier type="doi">10.63317/3kv2f8gmz5rb</identifier>
<location>
<url>https://aclanthology.org/2026.legal-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>53</start>
<end>61</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Distilling Human-Aligned Privacy Sensitivity Assessment from Large Language Models
%A Loiseau, Gabriel
%A Sileo, Damien
%A Riquet, Damien
%A Meyer, Maxime
%A Tommasi, Marc
%Y Siegert, Ingo
%Y Szawerna, Maria Irena
%Y Choukri, Khalid
%Y Dobnik, Simon
%Y Kamocki, Paweł
%Y Lindström Tiedemann, Therese
%Y Lison, Pierre
%Y Muñoz Sánchez, Ricardo
%Y Pilán, Ildikó
%Y Södergård, Lisa
%Y Talmoudi, Kossay
%Y Volodina, Elena
%Y Vu, Xuan-Son
%S Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy (LEGAL2026 and CALD-pseudo 2026) @ LREC 2026
%D 2026
%8 May
%I ELRA
%C Palma, Mallorca (Spain)
%F loiseau-etal-2026-distilling
%X Accurate privacy evaluation of textual data remains a critical challenge in privacy-preserving NLP. Recent work has shown that LLMs can serve as reliable privacy evaluators, achieving strong agreement with human judgments; however, their computational cost and impracticality for processing sensitive data at scale limit real-world deployment. We address this gap by distilling the privacy assessment capabilities of Mistral Large 3 (675B) into lightweight encoder models with as few as 150M parameters. Leveraging a large-scale dataset of privacy-annotated texts spanning 10 diverse domains, we train efficient classifiers that preserve strong agreement with human annotations while dramatically reducing computational requirements. We validate our approach on human-annotated test data and demonstrate its practical utility as an evaluation metric for de-identification systems.
%R 10.63317/3kv2f8gmz5rb
%U https://aclanthology.org/2026.legal-1.6/
%U https://doi.org/10.63317/3kv2f8gmz5rb
%P 53-61
Markdown (Informal)
[Distilling Human-Aligned Privacy Sensitivity Assessment from Large Language Models](https://aclanthology.org/2026.legal-1.6/) (Loiseau et al., LEGAL-CALD-pseudo 2026)
ACL
- Gabriel Loiseau, Damien Sileo, Damien Riquet, Maxime Meyer, and Marc Tommasi. 2026. Distilling Human-Aligned Privacy Sensitivity Assessment from Large Language Models. In Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy (LEGAL2026 and CALD-pseudo 2026) @ LREC 2026, pages 53–61, Palma, Mallorca (Spain). ELRA.