@InProceedings{arora-EtAl:2026:wmt1,
  author    = {Arora, Palak  and  Sharma, Adhikarimayum Meerajita  and  Indoria, Mrityunjaya  and  Lhoungu, Keneiwenuo  and  Lalthafamkimi, Lalthafamkimi  and  Nathani, Bharti  and  Joshi, Nisheeth},
  title     = {BVSLP: Enhancing Machine Translation through NMT Fine-Tuning, Back-Translation, and LLM-Assisted Synthetic Data Generation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2093--2100},
  abstract  = {This paper presents a two-stage neural machine translation framework for English and selected low-resource Northeastern Indian languages. The approach incorporated transfer learning, bidirectional fine-tuning, back-translation, and LLM-assisted synthetic data augmentation. In Stage 1, the NMT model was fine-tuned on cleaned gold-standard parallel data. In Stage 2, additional synthetic sentence pairs were generated using back-translation and Qwen2.5-Instruct, followed by filtering based on language identification, length ratio, duplicate removal, named-entity preservation, semantic similarity, and round-trip consistency. The framework was evaluated on English–Assamese, English–Mizo, English–Manipuri, English–Bodo, English–Khasi, and Nagamese–English translation directions. Results show stronger performance for several Northeastern-language-to-English directions, while the contrastive system notably improved Manipuri (Bengali)–English translation from 26.56 to 30.59 BLEU. The findings indicate that controlled synthetic augmentation can improve low-resource MT, although its effectiveness remains language- and direction-dependent.},
  url       = {https://aclanthology.org/2026.wmt-1.147}
}

