@inproceedings{L16-1358,
 abstract = {This paper describes a pilot study in lexical encoding of multi-word expressions (MWEs) in 4 Latin American dialects of Spanish: Costa Rican, Colombian, Mexican and Peruvian. We describe the variability of MWE usage across dialects. We adapt an existing data model to a dialect-aware encoding, so as to represent dialect-related specificities, while avoiding redundancy of the data common for all dialects. A dozen of linguistic properties of MWEs can be expressed in this model, both on the level of a whole MWE and of its individual components. We describe the resulting lexical resource containing several dozens of MWEs in four dialects and we propose a method for constructing a web corpus as a support for crowdsourcing examples of MWE occurrences. The resource is available under an open license and paves the way towards a large-scale dialect-aware language resource construction, which should prove useful in both traditional and novel NLP applications.
},
 address = {Portorož, Slovenia},
 author = {Diana Bogantes and Eric Rodríguez and Alejandro Arauco and Alejandro Rodríguez and Agata Savary},
 booktitle = {Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC 2016)},
 month = {May},
 pages = {2255--2261},
 publisher = {European Language Resources Association (ELRA)},
 title = {Towards Lexical Encoding of Multi-Word Expressions in Spanish Dialects},
 url = {https://www.aclweb.org/anthology/L16-1358},
 year = {2016}
}

