@inproceedings{kinouchi-etal-2026-layout,
title = "Layout-Based Chunk Alignment: Utilizing Visual Information to Collect Parallel Texts From Image Documents",
author = "Kinouchi, Masaki and
Nohara, Kayoko and
Zhu, Xinru and
Miura, Yuma",
editor = "Briakou, Eleftheria and
Gwinnup, Jeremy and
Goel, Shivali",
booktitle = "Proceedings of the 17th Conference of the Association for Machine Translation in the {A}mericas (Volume 1: Research Track)",
month = aug,
year = "2026",
address = "Qu{\'e}bec City, Canada",
publisher = "Association for Machine Translation in the Americas",
url = "https://aclanthology.org/2026.amta-research.8/",
pages = "135--145",
abstract = "This study proposes a layout-based chunk alignment method (Layout-CA) as an intermediate step between document- and sentence-level parallel text alignment for bilingual document images. Visually rich printed materials, such as institutional reports and magazines, often contain high-quality translations and are valuable sources of parallel data, yet their layout cues are underutilized. Layout-CA aligns semantically coherent text chunks across document pairs by integrating multi-modal cues from textual content and layout, and sentence alignment is then performed within the aligned chunk pairs. Experiments on English UNESCO reports and their Japanese translations show that introducing chunk alignment improves downstream sentence alignment for both Bleualign and Vecalign. When document order is disrupted, Layout-CA preserves alignment coverage by restricting sentence matching to corresponding chunks, enabling robust alignment in multilingual image documents."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kinouchi-etal-2026-layout">
<titleInfo>
<title>Layout-Based Chunk Alignment: Utilizing Visual Information to Collect Parallel Texts From Image Documents</title>
</titleInfo>
<name type="personal">
<namePart type="given">Masaki</namePart>
<namePart type="family">Kinouchi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kayoko</namePart>
<namePart type="family">Nohara</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xinru</namePart>
<namePart type="family">Zhu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yuma</namePart>
<namePart type="family">Miura</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Conference of the Association for Machine Translation in the Americas (Volume 1: Research Track)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Eleftheria</namePart>
<namePart type="family">Briakou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jeremy</namePart>
<namePart type="family">Gwinnup</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shivali</namePart>
<namePart type="family">Goel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Machine Translation in the Americas</publisher>
<place>
<placeTerm type="text">Québec City, Canada</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This study proposes a layout-based chunk alignment method (Layout-CA) as an intermediate step between document- and sentence-level parallel text alignment for bilingual document images. Visually rich printed materials, such as institutional reports and magazines, often contain high-quality translations and are valuable sources of parallel data, yet their layout cues are underutilized. Layout-CA aligns semantically coherent text chunks across document pairs by integrating multi-modal cues from textual content and layout, and sentence alignment is then performed within the aligned chunk pairs. Experiments on English UNESCO reports and their Japanese translations show that introducing chunk alignment improves downstream sentence alignment for both Bleualign and Vecalign. When document order is disrupted, Layout-CA preserves alignment coverage by restricting sentence matching to corresponding chunks, enabling robust alignment in multilingual image documents.</abstract>
<identifier type="citekey">kinouchi-etal-2026-layout</identifier>
<location>
<url>https://aclanthology.org/2026.amta-research.8/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>135</start>
<end>145</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Layout-Based Chunk Alignment: Utilizing Visual Information to Collect Parallel Texts From Image Documents
%A Kinouchi, Masaki
%A Nohara, Kayoko
%A Zhu, Xinru
%A Miura, Yuma
%Y Briakou, Eleftheria
%Y Gwinnup, Jeremy
%Y Goel, Shivali
%S Proceedings of the 17th Conference of the Association for Machine Translation in the Americas (Volume 1: Research Track)
%D 2026
%8 August
%I Association for Machine Translation in the Americas
%C Québec City, Canada
%F kinouchi-etal-2026-layout
%X This study proposes a layout-based chunk alignment method (Layout-CA) as an intermediate step between document- and sentence-level parallel text alignment for bilingual document images. Visually rich printed materials, such as institutional reports and magazines, often contain high-quality translations and are valuable sources of parallel data, yet their layout cues are underutilized. Layout-CA aligns semantically coherent text chunks across document pairs by integrating multi-modal cues from textual content and layout, and sentence alignment is then performed within the aligned chunk pairs. Experiments on English UNESCO reports and their Japanese translations show that introducing chunk alignment improves downstream sentence alignment for both Bleualign and Vecalign. When document order is disrupted, Layout-CA preserves alignment coverage by restricting sentence matching to corresponding chunks, enabling robust alignment in multilingual image documents.
%U https://aclanthology.org/2026.amta-research.8/
%P 135-145
Markdown (Informal)
[Layout-Based Chunk Alignment: Utilizing Visual Information to Collect Parallel Texts From Image Documents](https://aclanthology.org/2026.amta-research.8/) (Kinouchi et al., AMTA 2026)
ACL