@InProceedings{hede-EtAl:2026:wmt1,
  author    = {Hede, Vedarth  and  Gupta, Neha  and  Bapat, Harish  and  Ekbote, Harsh},
  title     = {MTG-LIRA: Scaling Neural Machine Translation for Low-Resource Indian Languages through Multilingual Learning and Synthetic Data},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2153--2160},
  abstract  = {MTG-LIRA's submission for the WMT 2026 Low-Resource Indic Language Translation Shared Task presents a unified, scalable multilingual Transformer architecture developed using OpenNMT-py to address severe data scarcity across several low-resource Indian languages. The framework partitions languages into family-based groupings—such as Indo-Aryan and Tibeto-Burman—to optimize cross-lingual knowledge sharing and vocabulary overlap using Byte Pair Encoding (BPE). Data preparation integrates official WMT datasets, the BPCC multilingual corpus, and ~5 million synthetic English–Mizo sentence pairs generated via Meta's NLLB model, all cleaned using a rigorous preprocessing pipeline. Experimental results across task directions demonstrate that combining multilingual base models with targeted, language-specific adaptation strategies yields significant gains: fine-tuning on domain-specific data improved BLEU scores for English→Assamese (12.17 to 14.44) and English→Mizo (16.2 to 22.72); checkpoint averaging of the 5 best models enhanced stability for English↔Manipuri (Meitei Mayek); a weighted 3:1 checkpoint averaging strategy improved performance for English→Bodo; and zero-shot style direct transfer proved effective for Assamese→English and English→Manipuri (Bengali script) due to shared script features.},
  url       = {https://aclanthology.org/2026.wmt-1.153}
}

