@article{yifei-etal-2026-r,
title = "{R} {ESEARCH} {QA}: Evaluating Scholarly Question Answering at Scale Across 75 Fields with Survey-Mined Questions and Rubrics",
author = "Yifei, Li S. and
Chang, Allen and
Malaviya, Chaitanya and
Yatskar, Mark",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.62/",
doi = "10.1162/tacl.a.732",
pages = "1365--1389",
abstract = "Evaluating long-form responses to research queries is increasingly important for LLM agents, particularly emerging deep research systems. Such evaluation heavily relies on expert annotators, restricting attention to areas like AI where researchers can conveniently enlist colleagues. Yet, research expertise is abundant: survey articles consolidate knowledge spread across the literature. We introduce RESEARCHQA, a resource for evaluating LLM systems by distilling survey articles from 75 research felds into 21K queries and 160K rubric items. Queries and rubrics are jointly derived from survey sections, where rubric items list query-specific answer evaluation criteria, i.e., citing papers, making explanations, and describing limitations. 31 Ph.D. annotators in 8 fields judge that 90{\%} of queries reflect Ph.D. information needs and 87{\%} of rubric items warrant emphasis of a sentence or longer. We leverage RESEARCHQA to evaluate 18 systems in 7.6K head-to-heads. No parametric or retrieval-augmented system we evaluate exceeds 70{\%} on covering rubric items, and the highest-ranking system shows 75{\%} coverage. Error analysis reveals that the highest-ranking system fully addresses less than 11{\%} of citation rubric items, 48{\%} of limitation items, and 49{\%} of comparison items. We release our data to facilitate more comprehensive multi-field evaluations."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="yifei-etal-2026-r">
<titleInfo>
<title>R ESEARCH QA: Evaluating Scholarly Question Answering at Scale Across 75 Fields with Survey-Mined Questions and Rubrics</title>
</titleInfo>
<name type="personal">
<namePart type="given">Li</namePart>
<namePart type="given">S</namePart>
<namePart type="family">Yifei</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Allen</namePart>
<namePart type="family">Chang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chaitanya</namePart>
<namePart type="family">Malaviya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mark</namePart>
<namePart type="family">Yatskar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Evaluating long-form responses to research queries is increasingly important for LLM agents, particularly emerging deep research systems. Such evaluation heavily relies on expert annotators, restricting attention to areas like AI where researchers can conveniently enlist colleagues. Yet, research expertise is abundant: survey articles consolidate knowledge spread across the literature. We introduce RESEARCHQA, a resource for evaluating LLM systems by distilling survey articles from 75 research felds into 21K queries and 160K rubric items. Queries and rubrics are jointly derived from survey sections, where rubric items list query-specific answer evaluation criteria, i.e., citing papers, making explanations, and describing limitations. 31 Ph.D. annotators in 8 fields judge that 90% of queries reflect Ph.D. information needs and 87% of rubric items warrant emphasis of a sentence or longer. We leverage RESEARCHQA to evaluate 18 systems in 7.6K head-to-heads. No parametric or retrieval-augmented system we evaluate exceeds 70% on covering rubric items, and the highest-ranking system shows 75% coverage. Error analysis reveals that the highest-ranking system fully addresses less than 11% of citation rubric items, 48% of limitation items, and 49% of comparison items. We release our data to facilitate more comprehensive multi-field evaluations.</abstract>
<identifier type="citekey">yifei-etal-2026-r</identifier>
<identifier type="doi">10.1162/tacl.a.732</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.62/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>1365</start>
<end>1389</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T R ESEARCH QA: Evaluating Scholarly Question Answering at Scale Across 75 Fields with Survey-Mined Questions and Rubrics
%A Yifei, Li S.
%A Chang, Allen
%A Malaviya, Chaitanya
%A Yatskar, Mark
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F yifei-etal-2026-r
%X Evaluating long-form responses to research queries is increasingly important for LLM agents, particularly emerging deep research systems. Such evaluation heavily relies on expert annotators, restricting attention to areas like AI where researchers can conveniently enlist colleagues. Yet, research expertise is abundant: survey articles consolidate knowledge spread across the literature. We introduce RESEARCHQA, a resource for evaluating LLM systems by distilling survey articles from 75 research felds into 21K queries and 160K rubric items. Queries and rubrics are jointly derived from survey sections, where rubric items list query-specific answer evaluation criteria, i.e., citing papers, making explanations, and describing limitations. 31 Ph.D. annotators in 8 fields judge that 90% of queries reflect Ph.D. information needs and 87% of rubric items warrant emphasis of a sentence or longer. We leverage RESEARCHQA to evaluate 18 systems in 7.6K head-to-heads. No parametric or retrieval-augmented system we evaluate exceeds 70% on covering rubric items, and the highest-ranking system shows 75% coverage. Error analysis reveals that the highest-ranking system fully addresses less than 11% of citation rubric items, 48% of limitation items, and 49% of comparison items. We release our data to facilitate more comprehensive multi-field evaluations.
%R 10.1162/tacl.a.732
%U https://aclanthology.org/2026.tacl-1.62/
%U https://doi.org/10.1162/tacl.a.732
%P 1365-1389
Markdown (Informal)
[R ESEARCH QA: Evaluating Scholarly Question Answering at Scale Across 75 Fields with Survey-Mined Questions and Rubrics](https://aclanthology.org/2026.tacl-1.62/) (Yifei et al., TACL 2026)
ACL