@InProceedings{ibrahim:2026:wmt,
  author    = {Ibrahim, Beshir},
  title     = {Closing the Resource Gap for Tigre: A Diaspora-Sourced English–Tigre Parallel Dataset for SMOL},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2509--2515},
  abstract  = {Tigre, an Afro-Semitic language maintaining strong linguistic continuity with Classical Ge'ez, holds a significant presence in Eritrea as well as across native and intergenerational refugee communities in eastern Sudan. It remains absent from major machine translation benchmarks and resources such as FLORES-200, despite an established orthography and active use in education. As a submission to the WMT26 Open Language Data shared task, we present a community-driven contribution of 11,998 English–Tigre pairs (word-, phrase-, and sentence-level) to the SMOL parallel corpus, written in the standardized Ge'ez-script form of Tigre used in Eritrean education and media. The data were produced through a structured, multi-stage post-editing and crossvalidation pipeline involving two diaspora native-speaker translator groups. We document the collection methodology and translator workflow, and validate the data's usefulness by fine-tuning NLLB-200-3.3B on the contributed pairs, observing about a 4.4× improvement in chrF++ over an untrained baseline for English-to-Tigre translation (4.92 to 21.63), with gains in both translation directions. We release the dataset under a Creative Commons Attribution (CC-BY 4.0) license to support future NLP research on this severely underresourced language},
  url       = {https://aclanthology.org/2026.wmt-1.195}
}

