@inproceedings{talmoudi-etal-2026-legal,
title = "Legal Considerations in the Use of Synthetic Data for {AI} Development and Finetuning: The Case of {LLM}s4{EU}",
author = "Talmoudi, Kossay and
Choukri, Khalid and
Gourgeot, Am{\'e}lie and
Astruc, Florine",
editor = {Siegert, Ingo and
Szawerna, Maria Irena and
Choukri, Khalid and
Dobnik, Simon and
Kamocki, Pawe{\l} and
Lindstr{\"o}m Tiedemann, Therese and
Lison, Pierre and
Mu{\~n}oz S{\'a}nchez, Ricardo and
Pil{\'a}n, Ildik{\'o} and
S{\"o}derg{\r{a}}rd, Lisa and
Talmoudi, Kossay and
Volodina, Elena and
Vu, Xuan-Son},
booktitle = "Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy ({LEGAL}2026 and {CALD}-pseudo 2026) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA",
url = "https://aclanthology.org/2026.legal-1.10/",
doi = "10.63317/3vwisn8odtmp",
pages = "86--90",
abstract = "This paper examines the legal implications of using synthetic data to develop and fine-tune general-purpose AI models in the European Union, using the LLMs4EU project as a case study. It situates synthetic data within the Union{'}s broader data policy and highlights it as a candidate tool for reconciling data availability with regulatory constraints. From a data-protection perspective, it analyses whether and when synthetic data should be classified as ``personal data'' under the GDPR. From a copyright and contractual standpoint, the paper assesses the risks that synthetic datasets may embed infringing content or derive from unlawfully trained models, in light of the GEMA v. OpenAI ruling on memorised works and emerging analyses of liability for AI-generated outputs, and considers the constraints imposed by model licensing and acceptable-use policies on using models to generate training data for other models. The paper concludes that synthetic data can play a valuable role in mitigating legal risks and enabling compliant AI development in LLMs4EU, but only if its generation and use are embedded in robust governance frameworks that address data protection, copyright and contractual obligations across the entire data value chain."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="talmoudi-etal-2026-legal">
<titleInfo>
<title>Legal Considerations in the Use of Synthetic Data for AI Development and Finetuning: The Case of LLMs4EU</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kossay</namePart>
<namePart type="family">Talmoudi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalid</namePart>
<namePart type="family">Choukri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amélie</namePart>
<namePart type="family">Gourgeot</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Florine</namePart>
<namePart type="family">Astruc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy (LEGAL2026 and CALD-pseudo 2026) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ingo</namePart>
<namePart type="family">Siegert</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="given">Irena</namePart>
<namePart type="family">Szawerna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalid</namePart>
<namePart type="family">Choukri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Dobnik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paweł</namePart>
<namePart type="family">Kamocki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Therese</namePart>
<namePart type="family">Lindström Tiedemann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Lison</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ricardo</namePart>
<namePart type="family">Muñoz Sánchez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ildikó</namePart>
<namePart type="family">Pilán</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Södergård</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kossay</namePart>
<namePart type="family">Talmoudi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Volodina</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xuan-Son</namePart>
<namePart type="family">Vu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper examines the legal implications of using synthetic data to develop and fine-tune general-purpose AI models in the European Union, using the LLMs4EU project as a case study. It situates synthetic data within the Union’s broader data policy and highlights it as a candidate tool for reconciling data availability with regulatory constraints. From a data-protection perspective, it analyses whether and when synthetic data should be classified as “personal data” under the GDPR. From a copyright and contractual standpoint, the paper assesses the risks that synthetic datasets may embed infringing content or derive from unlawfully trained models, in light of the GEMA v. OpenAI ruling on memorised works and emerging analyses of liability for AI-generated outputs, and considers the constraints imposed by model licensing and acceptable-use policies on using models to generate training data for other models. The paper concludes that synthetic data can play a valuable role in mitigating legal risks and enabling compliant AI development in LLMs4EU, but only if its generation and use are embedded in robust governance frameworks that address data protection, copyright and contractual obligations across the entire data value chain.</abstract>
<identifier type="citekey">talmoudi-etal-2026-legal</identifier>
<identifier type="doi">10.63317/3vwisn8odtmp</identifier>
<location>
<url>https://aclanthology.org/2026.legal-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>86</start>
<end>90</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Legal Considerations in the Use of Synthetic Data for AI Development and Finetuning: The Case of LLMs4EU
%A Talmoudi, Kossay
%A Choukri, Khalid
%A Gourgeot, Amélie
%A Astruc, Florine
%Y Siegert, Ingo
%Y Szawerna, Maria Irena
%Y Choukri, Khalid
%Y Dobnik, Simon
%Y Kamocki, Paweł
%Y Lindström Tiedemann, Therese
%Y Lison, Pierre
%Y Muñoz Sánchez, Ricardo
%Y Pilán, Ildikó
%Y Södergård, Lisa
%Y Talmoudi, Kossay
%Y Volodina, Elena
%Y Vu, Xuan-Son
%S Proceedings of the Joint Workshop on Legal and Ethical Issues in Human Language Technologies and Computational Approaches to Language Data Pseudonymization, Anonymization, De-identification, and Data Privacy (LEGAL2026 and CALD-pseudo 2026) @ LREC 2026
%D 2026
%8 May
%I ELRA
%C Palma, Mallorca (Spain)
%F talmoudi-etal-2026-legal
%X This paper examines the legal implications of using synthetic data to develop and fine-tune general-purpose AI models in the European Union, using the LLMs4EU project as a case study. It situates synthetic data within the Union’s broader data policy and highlights it as a candidate tool for reconciling data availability with regulatory constraints. From a data-protection perspective, it analyses whether and when synthetic data should be classified as “personal data” under the GDPR. From a copyright and contractual standpoint, the paper assesses the risks that synthetic datasets may embed infringing content or derive from unlawfully trained models, in light of the GEMA v. OpenAI ruling on memorised works and emerging analyses of liability for AI-generated outputs, and considers the constraints imposed by model licensing and acceptable-use policies on using models to generate training data for other models. The paper concludes that synthetic data can play a valuable role in mitigating legal risks and enabling compliant AI development in LLMs4EU, but only if its generation and use are embedded in robust governance frameworks that address data protection, copyright and contractual obligations across the entire data value chain.
%R 10.63317/3vwisn8odtmp
%U https://aclanthology.org/2026.legal-1.10/
%U https://doi.org/10.63317/3vwisn8odtmp
%P 86-90
Markdown (Informal)
[Legal Considerations in the Use of Synthetic Data for AI Development and Finetuning: The Case of LLMs4EU](https://aclanthology.org/2026.legal-1.10/) (Talmoudi et al., LEGAL-CALD-pseudo 2026)
ACL