@InProceedings{jana-EtAl:2026:wmt,
  author    = {Jana, Pramit  and  Adhikari, Arindam  and  Acharya, Priyobroto  and  Das, Dipankar},
  title     = {IndicMT at WMT 2026: Full Fine-Tuning of NLLB-200-600M with Round-Trip Filtering and Forward Self-Training for Arabic-Asian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2318--2324},
  abstract  = {We describe the IndicMT submission to the WMT 2026 Shared Task on Low-Resource Arabic-Asian Language Translation. Our system covers all ten translation directions between Arabic and Bangla, English, Hindi, Indonesian, and Urdu. For each direction, we fully fine-tune a separate NLLB-200-distilled-600M model using two NVIDIA T4 graphics processing units. The training pipeline combines Unicode normalization and length filtering with direction-specific round-trip consistency filtering of the available parallel data. We additionally augment each direction with 500 source-side sentences whose target translations are generated by the pretrained NLLB model, a procedure corresponding to forward self-training rather than conventional back-translation. On the official test set, our strongest relative results are obtained for Arabic$\rightarrow$Indonesian and Indonesian$\rightarrow$Arabic. Across all five language pairs, translation from Arabic obtains higher BLEU scores than translation into Arabic. Our results demonstrate that full adaptation of a compact multilingual translation model, combined with model-based data selection and synthetic-target augmentation, provides a practical baseline under limited computational resources.},
  url       = {https://aclanthology.org/2026.wmt-1.172}
}

