@article{shen-etal-2026-vertical,
title = "Vertical Routing: A Cost-Efficient Collaboration Routing Framework",
author = "Shen, Si and
Shen, Peijun and
Zhu, Danhao",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.104/",
doi = "10.1162/tacl.a.802",
pages = "2301--2317",
abstract = "Routing Large and Small Language Models (LLMs SLMs) is commonly framed as query-level difficulty prediction, yet lightweight routers are often unreliable and stronger evaluators introduce non-trivial overhead. We propose VERTICAL ROUTING, a stage-level collaboration framework that avoids monolithic difficulty prediction by allocating the large model to critical subtasks and delegating the remaining generation to a small model. VERTICAL ROUTING instantiates two templates: (i) domain-specific templates that decompose a task into ordered stages with criticality scores, and (ii) a robust default template that follows a prefix-first prior for general queries. Under a token budget, we allocate large-model capacity to the most critical stages, or to a budgeted prefix under the default template. Experiments show that VERTICAL ROUTING outperforms the strongest baseline (RouterDC) by 2.0 points in the averaged metric, while significantly reducing token usage by 48.9{\%} and large-model output share by 41.7{\%}. Additionally, it enhances stability by lowering the standard deviation by 87.5{\%}. These results highlight its advantages in accuracy, efficiency, and robustness.1"
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="shen-etal-2026-vertical">
<titleInfo>
<title>Vertical Routing: A Cost-Efficient Collaboration Routing Framework</title>
</titleInfo>
<name type="personal">
<namePart type="given">Si</namePart>
<namePart type="family">Shen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Peijun</namePart>
<namePart type="family">Shen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Danhao</namePart>
<namePart type="family">Zhu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Routing Large and Small Language Models (LLMs SLMs) is commonly framed as query-level difficulty prediction, yet lightweight routers are often unreliable and stronger evaluators introduce non-trivial overhead. We propose VERTICAL ROUTING, a stage-level collaboration framework that avoids monolithic difficulty prediction by allocating the large model to critical subtasks and delegating the remaining generation to a small model. VERTICAL ROUTING instantiates two templates: (i) domain-specific templates that decompose a task into ordered stages with criticality scores, and (ii) a robust default template that follows a prefix-first prior for general queries. Under a token budget, we allocate large-model capacity to the most critical stages, or to a budgeted prefix under the default template. Experiments show that VERTICAL ROUTING outperforms the strongest baseline (RouterDC) by 2.0 points in the averaged metric, while significantly reducing token usage by 48.9% and large-model output share by 41.7%. Additionally, it enhances stability by lowering the standard deviation by 87.5%. These results highlight its advantages in accuracy, efficiency, and robustness.1</abstract>
<identifier type="citekey">shen-etal-2026-vertical</identifier>
<identifier type="doi">10.1162/tacl.a.802</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.104/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>2301</start>
<end>2317</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Vertical Routing: A Cost-Efficient Collaboration Routing Framework
%A Shen, Si
%A Shen, Peijun
%A Zhu, Danhao
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F shen-etal-2026-vertical
%X Routing Large and Small Language Models (LLMs SLMs) is commonly framed as query-level difficulty prediction, yet lightweight routers are often unreliable and stronger evaluators introduce non-trivial overhead. We propose VERTICAL ROUTING, a stage-level collaboration framework that avoids monolithic difficulty prediction by allocating the large model to critical subtasks and delegating the remaining generation to a small model. VERTICAL ROUTING instantiates two templates: (i) domain-specific templates that decompose a task into ordered stages with criticality scores, and (ii) a robust default template that follows a prefix-first prior for general queries. Under a token budget, we allocate large-model capacity to the most critical stages, or to a budgeted prefix under the default template. Experiments show that VERTICAL ROUTING outperforms the strongest baseline (RouterDC) by 2.0 points in the averaged metric, while significantly reducing token usage by 48.9% and large-model output share by 41.7%. Additionally, it enhances stability by lowering the standard deviation by 87.5%. These results highlight its advantages in accuracy, efficiency, and robustness.1
%R 10.1162/tacl.a.802
%U https://aclanthology.org/2026.tacl-1.104/
%U https://doi.org/10.1162/tacl.a.802
%P 2301-2317
Markdown (Informal)
[Vertical Routing: A Cost-Efficient Collaboration Routing Framework](https://aclanthology.org/2026.tacl-1.104/) (Shen et al., TACL 2026)
ACL