@InProceedings{zafar-EtAl:2026:wmt,
  author    = {Zafar, Shomaiza  and  Abdul Rauf, Sadaf  and  Firdous, Sheema  and  Munir, Muhammad Saad  and  Fakhar, Nadeem},
  title     = {SLPG\_FJWU at WMT 2026: Domain Adapted Augmentation for Low-Resource Arabic-Asian Language Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2358--2366},
  abstract  = {This paper describes a submission to the WMT 2026 Low-Resource Arabic-Asian Language Translation shared task, covering three directions: Urdu→Arabic, Arabic→Urdu, and Arabic→English. For each direction we submit a primary system (official WMT data only) and a contrastive system trained on an augmented corpus. For the low resource Urdu-Arabic pair, we build a silver standard corpus by mining CC-100 Urdu monolingual text, applying domain adaptation to filter for relevance against WMT DevTest anchors using LEALLA sentence embeddings, and translating the retained sentences into Arabic — adding 27,172 domain matched pairs to the 20,299 official pairs. For Arabic-English, we augment the WMT data with News Commentary to expand the training pool roughly fivefold. All systems are built through transfer learning, fine-tuning the pretrained multilingual NLLB-200 model across all three directions. Our contrastive systems outperform our own primary systems in every direction (BLEU gains of 1.6–3.2 points), and on the official leaderboard our contrastive systems rank 1st in Arabic→Urdu, 3rd in Urdu→Arabic, and 5th in Arabic→English. Our results show that domain adaptation, rather than raw data volume, is the more effective lever for improving low resource NMT.},
  url       = {https://aclanthology.org/2026.wmt-1.178}
}

