@inproceedings{kiy-etal-2026-reproducibility,
title = "Reproducibility under Threat: Proposing a Framework for Reliable {LLM}-Research in Psychology and Computational Social Science",
author = "Kiy, Kevin Dirk and
Porshnev, Alexander and
Lakhzoum, Dounia and
Lynott, Dermot and
O{'}Donoghue, Diarmuid and
Singh, Manokamna",
editor = "Afli, Haithem and
Bouamor, Houda and
Zaghouani, Wajdi and
Ghannay, Sahar and
Hossain, Shehenaz",
booktitle = "Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences ({P}olitical{NLP} 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.politicalnlp-1.19/",
doi = "10.63317/2f5feg2mp6t3",
pages = "171--179",
abstract = "The integration of artificial intelligence (AI), particularly large language models (LLMs), into research across the social sciences has accelerated innovation but also introduced significant challenges to reproducibility - a cornerstone of scientific integrity. In this review of scientific practices, we examine the reproducibility crisis in AI-driven research with a focus on psychology, identifying common pitfalls, reviewing proposed solutions, and advocating for best practices. Common pitfalls in current practices in the social sciences are identified and highlighted through synthesized research scenarios, such as: (1) using inaccessible datasets or language models with restricted access, (2) treating black-box API outputs as stable observations ignoring updates and hidden changes, (3) producing single runs for measurements instead of stochastic draws for aggregated performances, (4) failing to report full LLM version, prompting, and sampling parameters, and (5) opaque training and fine-tuning of LLMs. Our recommended practices include precisely documenting the model used, fixing all inference parameters, using automation and scripts to control prompts, context, and outputs, and standardizing the environment and API conditions. By embracing transparency and methodological rigour, we can transform the challenges of AI-driven research into opportunities for more robust and impactful science, ensuring that innovation never comes at the cost of credibility."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kiy-etal-2026-reproducibility">
<titleInfo>
<title>Reproducibility under Threat: Proposing a Framework for Reliable LLM-Research in Psychology and Computational Social Science</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kevin</namePart>
<namePart type="given">Dirk</namePart>
<namePart type="family">Kiy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alexander</namePart>
<namePart type="family">Porshnev</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dounia</namePart>
<namePart type="family">Lakhzoum</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dermot</namePart>
<namePart type="family">Lynott</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Diarmuid</namePart>
<namePart type="family">O’Donoghue</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manokamna</namePart>
<namePart type="family">Singh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences (PoliticalNLP 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Haithem</namePart>
<namePart type="family">Afli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Houda</namePart>
<namePart type="family">Bouamor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wajdi</namePart>
<namePart type="family">Zaghouani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sahar</namePart>
<namePart type="family">Ghannay</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shehenaz</namePart>
<namePart type="family">Hossain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The integration of artificial intelligence (AI), particularly large language models (LLMs), into research across the social sciences has accelerated innovation but also introduced significant challenges to reproducibility - a cornerstone of scientific integrity. In this review of scientific practices, we examine the reproducibility crisis in AI-driven research with a focus on psychology, identifying common pitfalls, reviewing proposed solutions, and advocating for best practices. Common pitfalls in current practices in the social sciences are identified and highlighted through synthesized research scenarios, such as: (1) using inaccessible datasets or language models with restricted access, (2) treating black-box API outputs as stable observations ignoring updates and hidden changes, (3) producing single runs for measurements instead of stochastic draws for aggregated performances, (4) failing to report full LLM version, prompting, and sampling parameters, and (5) opaque training and fine-tuning of LLMs. Our recommended practices include precisely documenting the model used, fixing all inference parameters, using automation and scripts to control prompts, context, and outputs, and standardizing the environment and API conditions. By embracing transparency and methodological rigour, we can transform the challenges of AI-driven research into opportunities for more robust and impactful science, ensuring that innovation never comes at the cost of credibility.</abstract>
<identifier type="citekey">kiy-etal-2026-reproducibility</identifier>
<identifier type="doi">10.63317/2f5feg2mp6t3</identifier>
<location>
<url>https://aclanthology.org/2026.politicalnlp-1.19/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>171</start>
<end>179</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Reproducibility under Threat: Proposing a Framework for Reliable LLM-Research in Psychology and Computational Social Science
%A Kiy, Kevin Dirk
%A Porshnev, Alexander
%A Lakhzoum, Dounia
%A Lynott, Dermot
%A O’Donoghue, Diarmuid
%A Singh, Manokamna
%Y Afli, Haithem
%Y Bouamor, Houda
%Y Zaghouani, Wajdi
%Y Ghannay, Sahar
%Y Hossain, Shehenaz
%S Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences (PoliticalNLP 2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F kiy-etal-2026-reproducibility
%X The integration of artificial intelligence (AI), particularly large language models (LLMs), into research across the social sciences has accelerated innovation but also introduced significant challenges to reproducibility - a cornerstone of scientific integrity. In this review of scientific practices, we examine the reproducibility crisis in AI-driven research with a focus on psychology, identifying common pitfalls, reviewing proposed solutions, and advocating for best practices. Common pitfalls in current practices in the social sciences are identified and highlighted through synthesized research scenarios, such as: (1) using inaccessible datasets or language models with restricted access, (2) treating black-box API outputs as stable observations ignoring updates and hidden changes, (3) producing single runs for measurements instead of stochastic draws for aggregated performances, (4) failing to report full LLM version, prompting, and sampling parameters, and (5) opaque training and fine-tuning of LLMs. Our recommended practices include precisely documenting the model used, fixing all inference parameters, using automation and scripts to control prompts, context, and outputs, and standardizing the environment and API conditions. By embracing transparency and methodological rigour, we can transform the challenges of AI-driven research into opportunities for more robust and impactful science, ensuring that innovation never comes at the cost of credibility.
%R 10.63317/2f5feg2mp6t3
%U https://aclanthology.org/2026.politicalnlp-1.19/
%U https://doi.org/10.63317/2f5feg2mp6t3
%P 171-179
Markdown (Informal)
[Reproducibility under Threat: Proposing a Framework for Reliable LLM-Research in Psychology and Computational Social Science](https://aclanthology.org/2026.politicalnlp-1.19/) (Kiy et al., PoliticalNLP 2026)
ACL