@inproceedings{wonnink-etal-2026-improving,
title = "Improving Completeness in Deep Research Agents through Targeted Enrichment",
author = "Wonnink, Jesse and
Zavrel, Jakub and
Groth, Paul",
editor = "Rehm, Georg and
Dietze, Stefan and
Dessi, Danilo and
Maynard, Diana and
Schimmler, Sonja",
booktitle = "Proceedings of Natural Scientific Language Processing ({NSLP}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nslp-1.19/",
doi = "10.63317/5pjfdths5wey",
pages = "193--205",
abstract = "Deep research agents, AI systems that autonomously gather, synthesize, and report on complex topics, represent a significant advance in information synthesis, yet ensuring the completeness of their outputs remains an open challenge. A key bottleneck is query generation: current systems decompose research questions into subqueries via prompt engineering alone, offering no formal guarantees on diversity or coverage, which leads to redundant retrieval and gaps in the resulting reports. This paper presents HERO (High Enrichment Retrieval Orchestrator), a hierarchical deep research architecture that addresses this limitation through two complementary mechanisms. First, submodular optimization via a facility location objective provides mathematically grounded control over the relevance{--}diversity trade-off during query selection, replacing ad-hoc generation with provably diverse query sets. Second, a hierarchical enrichment stage independently analyzes each subquery pipeline{'}s intermediate synthesis for information gaps and issues targeted follow-up queries, enabling adaptive depth without cross-pipeline interference. We evaluate HERO across academic (ScholarQABench) and general-domain (DeepResearchGym) benchmarks. HERO achieves state-of-the-art coverage (Key Point Recall: 67.63), grounding (Citation F1: 91.57), and presentation quality on DeepResearchGym, and the highest scores on multi-paper synthesis tasks in ScholarQABench."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="wonnink-etal-2026-improving">
<titleInfo>
<title>Improving Completeness in Deep Research Agents through Targeted Enrichment</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jesse</namePart>
<namePart type="family">Wonnink</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jakub</namePart>
<namePart type="family">Zavrel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Groth</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stefan</namePart>
<namePart type="family">Dietze</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Danilo</namePart>
<namePart type="family">Dessi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Diana</namePart>
<namePart type="family">Maynard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sonja</namePart>
<namePart type="family">Schimmler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Deep research agents, AI systems that autonomously gather, synthesize, and report on complex topics, represent a significant advance in information synthesis, yet ensuring the completeness of their outputs remains an open challenge. A key bottleneck is query generation: current systems decompose research questions into subqueries via prompt engineering alone, offering no formal guarantees on diversity or coverage, which leads to redundant retrieval and gaps in the resulting reports. This paper presents HERO (High Enrichment Retrieval Orchestrator), a hierarchical deep research architecture that addresses this limitation through two complementary mechanisms. First, submodular optimization via a facility location objective provides mathematically grounded control over the relevance–diversity trade-off during query selection, replacing ad-hoc generation with provably diverse query sets. Second, a hierarchical enrichment stage independently analyzes each subquery pipeline’s intermediate synthesis for information gaps and issues targeted follow-up queries, enabling adaptive depth without cross-pipeline interference. We evaluate HERO across academic (ScholarQABench) and general-domain (DeepResearchGym) benchmarks. HERO achieves state-of-the-art coverage (Key Point Recall: 67.63), grounding (Citation F1: 91.57), and presentation quality on DeepResearchGym, and the highest scores on multi-paper synthesis tasks in ScholarQABench.</abstract>
<identifier type="citekey">wonnink-etal-2026-improving</identifier>
<identifier type="doi">10.63317/5pjfdths5wey</identifier>
<location>
<url>https://aclanthology.org/2026.nslp-1.19/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>193</start>
<end>205</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Improving Completeness in Deep Research Agents through Targeted Enrichment
%A Wonnink, Jesse
%A Zavrel, Jakub
%A Groth, Paul
%Y Rehm, Georg
%Y Dietze, Stefan
%Y Dessi, Danilo
%Y Maynard, Diana
%Y Schimmler, Sonja
%S Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F wonnink-etal-2026-improving
%X Deep research agents, AI systems that autonomously gather, synthesize, and report on complex topics, represent a significant advance in information synthesis, yet ensuring the completeness of their outputs remains an open challenge. A key bottleneck is query generation: current systems decompose research questions into subqueries via prompt engineering alone, offering no formal guarantees on diversity or coverage, which leads to redundant retrieval and gaps in the resulting reports. This paper presents HERO (High Enrichment Retrieval Orchestrator), a hierarchical deep research architecture that addresses this limitation through two complementary mechanisms. First, submodular optimization via a facility location objective provides mathematically grounded control over the relevance–diversity trade-off during query selection, replacing ad-hoc generation with provably diverse query sets. Second, a hierarchical enrichment stage independently analyzes each subquery pipeline’s intermediate synthesis for information gaps and issues targeted follow-up queries, enabling adaptive depth without cross-pipeline interference. We evaluate HERO across academic (ScholarQABench) and general-domain (DeepResearchGym) benchmarks. HERO achieves state-of-the-art coverage (Key Point Recall: 67.63), grounding (Citation F1: 91.57), and presentation quality on DeepResearchGym, and the highest scores on multi-paper synthesis tasks in ScholarQABench.
%R 10.63317/5pjfdths5wey
%U https://aclanthology.org/2026.nslp-1.19/
%U https://doi.org/10.63317/5pjfdths5wey
%P 193-205
Markdown (Informal)
[Improving Completeness in Deep Research Agents through Targeted Enrichment](https://aclanthology.org/2026.nslp-1.19/) (Wonnink et al., NSLP 2026)
ACL