@inproceedings{brisson-etal-2026-anandasky,
title = "{A}nanda{S}ky: A Vision{--}Language Model for Line-Level Transcription of Historical Sinographic Documents",
author = "Brisson, Colin and
Kahfy, Ayoub and
Constant, Fr{\'e}d{\'e}ric and
Bui, Marc",
editor = "Sprugnoli, Rachele and
Passarotti, Marco",
booktitle = "Proceedings of the Fourth Workshop on Language Technologies for Historical and Ancient Languages ({LT}4{HALA} 2026) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.lt4hala-1.32/",
doi = "10.63317/3pk7cv8hxzod",
pages = "311--321",
abstract = "We present AnandaSky, a vision{--}language model for line-level transcription of historical sinographic documents. The model combines a compact high-resolution visual encoder with global attention, 10px patches, uncompressed visual prefix and a Qwen3-0.6B autoregressive decoder. It is trained at scale on 4M annotated lines from documents produced in China and Korea between the 8th and 20th centuries. Across in-domain and held-out public benchmarks, AnandaSky achieves sub-1{\%} CER on five of eight datasets, sets a new state of the art on MTHv2 with 0.92{\%} CER, and shows strong transfer to unseen collections. For EvaHan 2026, full fine-tuning on the organizers' data to match task-specific annotation conventions reduces CER relative to the official baseline by 5.2{\%} on prints and 12.1{\%} on manuscripts, despite using one-tenth as many parameters."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="brisson-etal-2026-anandasky">
<titleInfo>
<title>AnandaSky: A Vision–Language Model for Line-Level Transcription of Historical Sinographic Documents</title>
</titleInfo>
<name type="personal">
<namePart type="given">Colin</namePart>
<namePart type="family">Brisson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayoub</namePart>
<namePart type="family">Kahfy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Frédéric</namePart>
<namePart type="family">Constant</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Bui</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fourth Workshop on Language Technologies for Historical and Ancient Languages (LT4HALA 2026) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Rachele</namePart>
<namePart type="family">Sprugnoli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="family">Passarotti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present AnandaSky, a vision–language model for line-level transcription of historical sinographic documents. The model combines a compact high-resolution visual encoder with global attention, 10px patches, uncompressed visual prefix and a Qwen3-0.6B autoregressive decoder. It is trained at scale on 4M annotated lines from documents produced in China and Korea between the 8th and 20th centuries. Across in-domain and held-out public benchmarks, AnandaSky achieves sub-1% CER on five of eight datasets, sets a new state of the art on MTHv2 with 0.92% CER, and shows strong transfer to unseen collections. For EvaHan 2026, full fine-tuning on the organizers’ data to match task-specific annotation conventions reduces CER relative to the official baseline by 5.2% on prints and 12.1% on manuscripts, despite using one-tenth as many parameters.</abstract>
<identifier type="citekey">brisson-etal-2026-anandasky</identifier>
<identifier type="doi">10.63317/3pk7cv8hxzod</identifier>
<location>
<url>https://aclanthology.org/2026.lt4hala-1.32/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>311</start>
<end>321</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T AnandaSky: A Vision–Language Model for Line-Level Transcription of Historical Sinographic Documents
%A Brisson, Colin
%A Kahfy, Ayoub
%A Constant, Frédéric
%A Bui, Marc
%Y Sprugnoli, Rachele
%Y Passarotti, Marco
%S Proceedings of the Fourth Workshop on Language Technologies for Historical and Ancient Languages (LT4HALA 2026) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F brisson-etal-2026-anandasky
%X We present AnandaSky, a vision–language model for line-level transcription of historical sinographic documents. The model combines a compact high-resolution visual encoder with global attention, 10px patches, uncompressed visual prefix and a Qwen3-0.6B autoregressive decoder. It is trained at scale on 4M annotated lines from documents produced in China and Korea between the 8th and 20th centuries. Across in-domain and held-out public benchmarks, AnandaSky achieves sub-1% CER on five of eight datasets, sets a new state of the art on MTHv2 with 0.92% CER, and shows strong transfer to unseen collections. For EvaHan 2026, full fine-tuning on the organizers’ data to match task-specific annotation conventions reduces CER relative to the official baseline by 5.2% on prints and 12.1% on manuscripts, despite using one-tenth as many parameters.
%R 10.63317/3pk7cv8hxzod
%U https://aclanthology.org/2026.lt4hala-1.32/
%U https://doi.org/10.63317/3pk7cv8hxzod
%P 311-321
Markdown (Informal)
[AnandaSky: A Vision–Language Model for Line-Level Transcription of Historical Sinographic Documents](https://aclanthology.org/2026.lt4hala-1.32/) (Brisson et al., LT4HALA 2026)
ACL