@inproceedings{stephens-etal-2026-large,
title = "Can Large Language Models Facilitate Qualitative Political Narrative Analysis?",
author = "Stephens, Luke and
Llewellyn, Clare and
Rogers, Lauren and
Kyritsopoulos, Constantine and
Prangere, Arman and
Long, Feiteng and
Snyder, Peyton and
Cram, Laura",
editor = "Afli, Haithem and
Bouamor, Houda and
Zaghouani, Wajdi and
Ghannay, Sahar and
Hossain, Shehenaz",
booktitle = "Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences ({P}olitical{NLP} 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.politicalnlp-1.28/",
doi = "10.63317/4rwv2gsphhck",
pages = "261--271",
abstract = "This study evaluates whether Large Language Models (LLMs) can facilitate qualitative political narrative analysis by comparing outputs from four models{---}Mistral, Llama, ChatGPT-4o, and DeepSeek{---}against narrative analyses written by expert scholars. Using European Union State of the Union speeches (2010{--}2023), we examine migration and solidarity narratives through semantic and lexical similarity metrics alongside systematic validation. The narrative scholars demonstrate strong semantic alignment despite differences in wording, establishing a benchmark for interpretive consistency. Across both topics, the models produce lexical and semantic similarity scores that are broadly comparable to those observed between the scholars themselves, with differences at these levels often marginal. However, similarity metrics do not provide the full picture. Validation reveals model-specific weaknesses that are not captured by lexical or semantic alignment alone, including factual errors, over-structural abstraction, and difficulty engaging less salient narrative threads. These findings demonstrate that LLMs can produce narratives that align closely with human outputs in semantic and lexical similarity, yet these measures alone are insufficient to assess interpretive quality."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="stephens-etal-2026-large">
<titleInfo>
<title>Can Large Language Models Facilitate Qualitative Political Narrative Analysis?</title>
</titleInfo>
<name type="personal">
<namePart type="given">Luke</namePart>
<namePart type="family">Stephens</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Clare</namePart>
<namePart type="family">Llewellyn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lauren</namePart>
<namePart type="family">Rogers</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Constantine</namePart>
<namePart type="family">Kyritsopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Arman</namePart>
<namePart type="family">Prangere</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Feiteng</namePart>
<namePart type="family">Long</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Peyton</namePart>
<namePart type="family">Snyder</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Laura</namePart>
<namePart type="family">Cram</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences (PoliticalNLP 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Haithem</namePart>
<namePart type="family">Afli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Houda</namePart>
<namePart type="family">Bouamor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wajdi</namePart>
<namePart type="family">Zaghouani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sahar</namePart>
<namePart type="family">Ghannay</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shehenaz</namePart>
<namePart type="family">Hossain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This study evaluates whether Large Language Models (LLMs) can facilitate qualitative political narrative analysis by comparing outputs from four models—Mistral, Llama, ChatGPT-4o, and DeepSeek—against narrative analyses written by expert scholars. Using European Union State of the Union speeches (2010–2023), we examine migration and solidarity narratives through semantic and lexical similarity metrics alongside systematic validation. The narrative scholars demonstrate strong semantic alignment despite differences in wording, establishing a benchmark for interpretive consistency. Across both topics, the models produce lexical and semantic similarity scores that are broadly comparable to those observed between the scholars themselves, with differences at these levels often marginal. However, similarity metrics do not provide the full picture. Validation reveals model-specific weaknesses that are not captured by lexical or semantic alignment alone, including factual errors, over-structural abstraction, and difficulty engaging less salient narrative threads. These findings demonstrate that LLMs can produce narratives that align closely with human outputs in semantic and lexical similarity, yet these measures alone are insufficient to assess interpretive quality.</abstract>
<identifier type="citekey">stephens-etal-2026-large</identifier>
<identifier type="doi">10.63317/4rwv2gsphhck</identifier>
<location>
<url>https://aclanthology.org/2026.politicalnlp-1.28/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>261</start>
<end>271</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Can Large Language Models Facilitate Qualitative Political Narrative Analysis?
%A Stephens, Luke
%A Llewellyn, Clare
%A Rogers, Lauren
%A Kyritsopoulos, Constantine
%A Prangere, Arman
%A Long, Feiteng
%A Snyder, Peyton
%A Cram, Laura
%Y Afli, Haithem
%Y Bouamor, Houda
%Y Zaghouani, Wajdi
%Y Ghannay, Sahar
%Y Hossain, Shehenaz
%S Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences (PoliticalNLP 2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F stephens-etal-2026-large
%X This study evaluates whether Large Language Models (LLMs) can facilitate qualitative political narrative analysis by comparing outputs from four models—Mistral, Llama, ChatGPT-4o, and DeepSeek—against narrative analyses written by expert scholars. Using European Union State of the Union speeches (2010–2023), we examine migration and solidarity narratives through semantic and lexical similarity metrics alongside systematic validation. The narrative scholars demonstrate strong semantic alignment despite differences in wording, establishing a benchmark for interpretive consistency. Across both topics, the models produce lexical and semantic similarity scores that are broadly comparable to those observed between the scholars themselves, with differences at these levels often marginal. However, similarity metrics do not provide the full picture. Validation reveals model-specific weaknesses that are not captured by lexical or semantic alignment alone, including factual errors, over-structural abstraction, and difficulty engaging less salient narrative threads. These findings demonstrate that LLMs can produce narratives that align closely with human outputs in semantic and lexical similarity, yet these measures alone are insufficient to assess interpretive quality.
%R 10.63317/4rwv2gsphhck
%U https://aclanthology.org/2026.politicalnlp-1.28/
%U https://doi.org/10.63317/4rwv2gsphhck
%P 261-271
Markdown (Informal)
[Can Large Language Models Facilitate Qualitative Political Narrative Analysis?](https://aclanthology.org/2026.politicalnlp-1.28/) (Stephens et al., PoliticalNLP 2026)
ACL
- Luke Stephens, Clare Llewellyn, Lauren Rogers, Constantine Kyritsopoulos, Arman Prangere, Feiteng Long, Peyton Snyder, and Laura Cram. 2026. Can Large Language Models Facilitate Qualitative Political Narrative Analysis?. In Proceedings of the 3rd Workshop on Natural Language Processing for Political Sciences (PoliticalNLP 2026), pages 261–271, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).