@inproceedings{yung-etal-2026-semi,
title = "Semi-automatic Approach for {T}amil Discourse Relation Annotation",
author = "Yung, Frances and
Ponraj, Enosh Peter and
Demberg, Vera",
editor = "Jha, Girish Nath and
Bali, Kalika and
L, Sobha and
Kumar, Devendr",
booktitle = "Proceedings of the 8th Workshop on {I}ndian Language Data: Resources and Evaluation",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.wildre-1.2/",
doi = "10.63317/2imw3didyv2n",
pages = "14--24",
abstract = "Discourse relations (DRs) specify the logical relations between text spans and are essential for modeling extended discourse. Resources annotated with DRs can help train large language models (LLMs) to recognize and generate these relations more naturally. However, there is currently no open-source DR-annotated resource for Tamil. Annotation is particularly challenging because many Tamil discourse connectives are realized as morphologically complex suffixes rather than standalone tokens, often involving phonological alternations. In this work, we present a DR-annotated dataset for Tamil based on the PDTB framework. We adopt a semi-automatic pipeline: 1) projection of automatic English discourse annotations onto Tamil in a parallel corpus; 2) lexical normalization using a morphological analyzer; and 3) manual verification of each instance. The resulting resource contains approximately 7;200 explicit DR annotations and a lexicon of 450 Tamil discourse connectives. The annotated data is available for download at https://anonymous.4open.science/r/Tamil-Semi-Automatic-Discourse-Relation-Dataset/."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="yung-etal-2026-semi">
<titleInfo>
<title>Semi-automatic Approach for Tamil Discourse Relation Annotation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Frances</namePart>
<namePart type="family">Yung</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Enosh</namePart>
<namePart type="given">Peter</namePart>
<namePart type="family">Ponraj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vera</namePart>
<namePart type="family">Demberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 8th Workshop on Indian Language Data: Resources and Evaluation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Girish</namePart>
<namePart type="given">Nath</namePart>
<namePart type="family">Jha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kalika</namePart>
<namePart type="family">Bali</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sobha</namePart>
<namePart type="family">L</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Devendr</namePart>
<namePart type="family">Kumar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Discourse relations (DRs) specify the logical relations between text spans and are essential for modeling extended discourse. Resources annotated with DRs can help train large language models (LLMs) to recognize and generate these relations more naturally. However, there is currently no open-source DR-annotated resource for Tamil. Annotation is particularly challenging because many Tamil discourse connectives are realized as morphologically complex suffixes rather than standalone tokens, often involving phonological alternations. In this work, we present a DR-annotated dataset for Tamil based on the PDTB framework. We adopt a semi-automatic pipeline: 1) projection of automatic English discourse annotations onto Tamil in a parallel corpus; 2) lexical normalization using a morphological analyzer; and 3) manual verification of each instance. The resulting resource contains approximately 7;200 explicit DR annotations and a lexicon of 450 Tamil discourse connectives. The annotated data is available for download at https://anonymous.4open.science/r/Tamil-Semi-Automatic-Discourse-Relation-Dataset/.</abstract>
<identifier type="citekey">yung-etal-2026-semi</identifier>
<identifier type="doi">10.63317/2imw3didyv2n</identifier>
<location>
<url>https://aclanthology.org/2026.wildre-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>14</start>
<end>24</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Semi-automatic Approach for Tamil Discourse Relation Annotation
%A Yung, Frances
%A Ponraj, Enosh Peter
%A Demberg, Vera
%Y Jha, Girish Nath
%Y Bali, Kalika
%Y L, Sobha
%Y Kumar, Devendr
%S Proceedings of the 8th Workshop on Indian Language Data: Resources and Evaluation
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca, Spain
%F yung-etal-2026-semi
%X Discourse relations (DRs) specify the logical relations between text spans and are essential for modeling extended discourse. Resources annotated with DRs can help train large language models (LLMs) to recognize and generate these relations more naturally. However, there is currently no open-source DR-annotated resource for Tamil. Annotation is particularly challenging because many Tamil discourse connectives are realized as morphologically complex suffixes rather than standalone tokens, often involving phonological alternations. In this work, we present a DR-annotated dataset for Tamil based on the PDTB framework. We adopt a semi-automatic pipeline: 1) projection of automatic English discourse annotations onto Tamil in a parallel corpus; 2) lexical normalization using a morphological analyzer; and 3) manual verification of each instance. The resulting resource contains approximately 7;200 explicit DR annotations and a lexicon of 450 Tamil discourse connectives. The annotated data is available for download at https://anonymous.4open.science/r/Tamil-Semi-Automatic-Discourse-Relation-Dataset/.
%R 10.63317/2imw3didyv2n
%U https://aclanthology.org/2026.wildre-1.2/
%U https://doi.org/10.63317/2imw3didyv2n
%P 14-24
Markdown (Informal)
[Semi-automatic Approach for Tamil Discourse Relation Annotation](https://aclanthology.org/2026.wildre-1.2/) (Yung et al., WILDRE 2026)
ACL