@inproceedings{kosai-etal-2026-llm,
title = "{LLM}-as-a-Judge Evaluation of Financial News Articles Generated Based on Factors of Stock Price Fluctuation",
author = "Kosai, Yurina and
Xie, Yucheng and
Tsuchida, Rikuto and
Utsuro, Takehito",
editor = "El-Haj, Mo and
Moreno Sandoval, Antonio and
Garcia-Serrano, Ana and
Chen, Chung-Chi and
Rayson, Paul and
Torterolo Orta, Yanco Amor and
Martinez, Paloma and
Porta, Jordi",
booktitle = "The 7th Financial Narrative Processing Workshop",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "European Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.fnp-1.10/",
doi = "10.63317/3axxq3b2dynj",
pages = "106--113",
abstract = "This paper proposes an LLM-as-a-Judge evaluation framework of stock price fluctuation articles automatically generated based on financial news, corporate disclosures, and stock price fluctuation data. This automatic article generation framework emulates the workflow of human financial journalists by analyzing recent stock price fluctuations and incorporating relevant causal factors extracted from textual and numerical information. In particular, the generation process utilizes news articles and numerical stock price data, including price fluctuation ranges over the past three days. Based on those automatically generated stock price fluctuation articles, this study places particular emphasis on the LLM-as-a-Judge evaluation methodology. We conduct an item wise human evaluation and compare it with the LLM-as-a-Judge automatic metric. We analyze the correlation among these evaluation methods to assess their reliability. Furthermore, through comparisons between zero-shot and few-shot prompting, we examine the effectiveness of the proposed framework and the validity of LLM based evaluation for assessing factual and causal consistency in financial text generation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kosai-etal-2026-llm">
<titleInfo>
<title>LLM-as-a-Judge Evaluation of Financial News Articles Generated Based on Factors of Stock Price Fluctuation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Yurina</namePart>
<namePart type="family">Kosai</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yucheng</namePart>
<namePart type="family">Xie</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rikuto</namePart>
<namePart type="family">Tsuchida</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Takehito</namePart>
<namePart type="family">Utsuro</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Financial Narrative Processing Workshop</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Moreno Sandoval</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ana</namePart>
<namePart type="family">Garcia-Serrano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chung-Chi</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Rayson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yanco</namePart>
<namePart type="given">Amor</namePart>
<namePart type="family">Torterolo Orta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paloma</namePart>
<namePart type="family">Martinez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jordi</namePart>
<namePart type="family">Porta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper proposes an LLM-as-a-Judge evaluation framework of stock price fluctuation articles automatically generated based on financial news, corporate disclosures, and stock price fluctuation data. This automatic article generation framework emulates the workflow of human financial journalists by analyzing recent stock price fluctuations and incorporating relevant causal factors extracted from textual and numerical information. In particular, the generation process utilizes news articles and numerical stock price data, including price fluctuation ranges over the past three days. Based on those automatically generated stock price fluctuation articles, this study places particular emphasis on the LLM-as-a-Judge evaluation methodology. We conduct an item wise human evaluation and compare it with the LLM-as-a-Judge automatic metric. We analyze the correlation among these evaluation methods to assess their reliability. Furthermore, through comparisons between zero-shot and few-shot prompting, we examine the effectiveness of the proposed framework and the validity of LLM based evaluation for assessing factual and causal consistency in financial text generation.</abstract>
<identifier type="citekey">kosai-etal-2026-llm</identifier>
<identifier type="doi">10.63317/3axxq3b2dynj</identifier>
<location>
<url>https://aclanthology.org/2026.fnp-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>106</start>
<end>113</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LLM-as-a-Judge Evaluation of Financial News Articles Generated Based on Factors of Stock Price Fluctuation
%A Kosai, Yurina
%A Xie, Yucheng
%A Tsuchida, Rikuto
%A Utsuro, Takehito
%Y El-Haj, Mo
%Y Moreno Sandoval, Antonio
%Y Garcia-Serrano, Ana
%Y Chen, Chung-Chi
%Y Rayson, Paul
%Y Torterolo Orta, Yanco Amor
%Y Martinez, Paloma
%Y Porta, Jordi
%S The 7th Financial Narrative Processing Workshop
%D 2026
%8 May
%I European Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F kosai-etal-2026-llm
%X This paper proposes an LLM-as-a-Judge evaluation framework of stock price fluctuation articles automatically generated based on financial news, corporate disclosures, and stock price fluctuation data. This automatic article generation framework emulates the workflow of human financial journalists by analyzing recent stock price fluctuations and incorporating relevant causal factors extracted from textual and numerical information. In particular, the generation process utilizes news articles and numerical stock price data, including price fluctuation ranges over the past three days. Based on those automatically generated stock price fluctuation articles, this study places particular emphasis on the LLM-as-a-Judge evaluation methodology. We conduct an item wise human evaluation and compare it with the LLM-as-a-Judge automatic metric. We analyze the correlation among these evaluation methods to assess their reliability. Furthermore, through comparisons between zero-shot and few-shot prompting, we examine the effectiveness of the proposed framework and the validity of LLM based evaluation for assessing factual and causal consistency in financial text generation.
%R 10.63317/3axxq3b2dynj
%U https://aclanthology.org/2026.fnp-1.10/
%U https://doi.org/10.63317/3axxq3b2dynj
%P 106-113
Markdown (Informal)
[LLM-as-a-Judge Evaluation of Financial News Articles Generated Based on Factors of Stock Price Fluctuation](https://aclanthology.org/2026.fnp-1.10/) (Kosai et al., FNP 2026)
ACL