@InProceedings{hede-EtAl:2026:wmt2,
  author    = {Hede, Vedarth  and  Gupta, Neha  and  Bapat, Harish  and  Ekbote, Harsh},
  title     = {MTG-LIRA at WMT 2026: Data-Centric Arabic-Hindi Machine Translation through Corpus Augmentation, Synthetic Parallel Data Generation and Transformer Models.},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2311--2317},
  abstract  = {MTG-LIRA's submission for the WMT 2026 Arabic–Asian Machine Translation Shared Task presents a data-centric Neural Machine Translation (NMT) pipeline to address low-resource Arabic–Hindi translation. Built using the OpenNMT-py framework with a 6-layer encoder-decoder Transformer architecture, the system relies on extensive corpus augmentation, combining official WMT data with ~4.05 million parallel sentence pairs from OPUS, alongside ~7 million synthetic Hindi–Arabic sentence pairs generated by translating the English side of the BPCC corpus (English-Hindi) into Arabic using Meta's NLLB model. The data undergoes rigorous preprocessing, including Unicode normalization, quality filtering, deduplication, and alignment verification, before evaluating both Byte Pair Encoding (BPE) and SentencePiece tokenization (32,000 vocabulary size) across two training strategies. Results show that while BPE and SentencePiece achieve comparable performance, the impact of synthetic data depends heavily on the translation direction: augmenting with synthetic data yields substantial performance improvements for Hindi→Arabic (increasing BLEU from 7.2 to 9.5 and COMET from 0.8166 to 0.8262 using BPE), whereas it degrades translation quality for Arabic→Hindi (where the OPUS-only BPE model achieves a higher BLEU of 25.3 compared to 17.7 with synthetic data).},
  url       = {https://aclanthology.org/2026.wmt-1.171}
}

